EADST

Pytorch Q4_1 Quantize and Dequantize aligning with llama.cpp

Pytorch Q4_1 Quantize and Dequantize aligning with llama.cpp

import torch

# Check if CUDA is available
use_cuda = torch.cuda.is_available()
device = torch.device("cuda" if use_cuda else "cpu")

def q4_1_quantize_and_dequantize_tensor(tensor):
    tensor = tensor.to(dtype=torch.float32, device=device)

    # Reshape tensor to process each 4-value block independently
    orig_shape = tensor.shape
    tensor = tensor.view(-1, 32)

    # Find the min and max values per block
    min_vals = torch.min(tensor, dim=1)[0]
    max_vals = torch.max(tensor, dim=1)[0]

    # Calculate scale d for each block
    d = (max_vals - min_vals) / (2**4 - 1)
    d[d == 0] = 1.0  # Prevent division by zero

    # Calculate inverse of d
    ids = 1.0 / d

    # Quantize tensor elements
    quantized_tensors = (tensor - min_vals[:, None]) * ids[:, None]

    # Clamp values to be between 0 and 15 (for 4 bits)
    quantized_tensors = torch.clamp(quantized_tensors + 0.5, 0, 15).to(torch.uint8)

    # Dequantize the tensor
    dequantized_tensors = (quantized_tensors.float() * d[:, None]) + min_vals[:, None]

    # Reshape back to the original shape
    dequantized_tensors = dequantized_tensors.view(orig_shape).to(dtype=torch.float16)

    return dequantized_tensors

# Assuming 'model_part' is already loaded and on CPU
model_part = torch.load(f"your_model_path/pytorch_model.bin", map_location="cpu")
keywords = [
    "embed_tokens.weight",
    "self_attn.q_proj.weight",
    "self_attn.k_proj.weight",
    "self_attn.v_proj.weight",
    "self_attn.o_proj.weight",
    "mlp.up_proj.weight",
    "mlp.gate_proj.weight",
    "mlp.down_proj.weight",
    "lm_head.weight"
]
for name, data in model_part.items():
    for word in keywords:
        if word in name:
            # Quantize and dequantize the entire tensor
            model_part[name] = q4_1_quantize_and_dequantize_tensor(data)

# Save the updated model parts
torch.save(model_part, "pytorch_model_quantized.bin")

Reference:

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
GPTQ ModelScope FastAPI Magnet FP32 tar 音频 Qwen2.5 DeepSeek 云服务器 PIP v2ray Numpy PDB SQL Algorithm AI Statistics Paddle 图标 Nginx Image2Text git-lfs CTC Ubuntu LLM GGML CLAP Jetson 继承 PyTorch Hungarian NLP 域名 Agent VGG-16 Jupyter TTS LoRA RGB Qwen2 Python Dataset 第一性原理 Tracking v0.dev Domain transformers Bipartite API Quantization Conda GIT 关于博主 Website Disk Firewall Clash LLAMA TensorFlow CEIR OpenCV BTC 阿里云 CV C++ uWSGI Git uwsgi Mixtral Llama Rebuttal VPN Permission SQLite 腾讯云 图形思考法 ONNX SPIE UNIX Vmess hf LeetCode FP64 CSV Pillow Math Interview Input FlashAttention QWEN IndexTTS2 Github 搞笑 Cloudreve 证件照 CUDA PDF 论文速读 YOLO Freesound Markdown Streamlit Pandas FP8 NLTK Hotel Pickle COCO Video 算法题 递归学习法 飞书 Bert Datetime VSCode WAN Michelin Qwen Linux Logo OpenAI DeepStream Heatmap Color XML XGBoost Proxy Ptyhon llama.cpp Shortcut InvalidArgumentError GoogLeNet 版权 tqdm logger Claude 多进程 Bin Animate CAM Translation Bitcoin Tiktoken BeautifulSoup ms-swift HuggingFace Excel Base64 OCR Distillation JSON GPT4 Knowledge Django MD5 mmap icon Web Use Docker WebCrawler EXCEL UI Password diffusers Food Tensor 签证 TensorRT 净利润 SAM scipy torchinfo Quantize 顶会 Pytorch 报税 Gemma News CC Anaconda LaTeX Card Crawler SVR Augmentation ResNet-50 Template Baidu Attention Zip HaggingFace Breakpoint printf ChatGPT 财报 论文 FP16 Land Random Search Data RAR Plotly NameSilo Hilton Safetensors Vim git RL TSV Google Miniforge Transformers Sklearn Paper 强化学习 Windows Diagram BF16 公式 Plate Review PyCharm 多线程
站点统计

本站现有博文333篇,共被浏览913612

本站已经建立2616天!

热门文章
文章归档
回到顶部