EADST

Pytorch GPTQ Dequantizing Function

Pytorch GPTQ Dequantizing Function

Here is the Python code optimizing the dequantization of a GPTQ model to torch FP16 format.

import torch

# Function: Dequantize quantized weights
def dequantization(qweight, qzeros, scales, g_idx, bits=4, group_size=128, device='cuda:0'):
    # Create a tensor for bitwise right shift operation
    wf = torch.tensor(list(range(0, 32, bits)), dtype=torch.int32).unsqueeze(0)

    # Apply bitwise right shift and convert qzeros to the appropriate type
    zeros = torch.bitwise_right_shift(torch.unsqueeze(qzeros, 2).expand(-1, -1, 32 // bits), wf.unsqueeze(0)).to(torch.int16 if bits == 8 else torch.int8)
    torch.bitwise_and(zeros, (2 ** bits) - 1, out=zeros)

    # Reshape the zeros tensor
    zeros = zeros + 1
    zeros = zeros.reshape(-1, 1, zeros.shape[1] * zeros.shape[2])

    # Reshape the scales tensor
    scales = scales.reshape(-1, 1, scales.shape[-1])

    # Similar bitwise right shift operation for qweight and reshape
    weight = torch.bitwise_right_shift(torch.unsqueeze(qweight, 1).expand(-1, 32 // bits, -1), wf.unsqueeze(-1)).to(torch.int16 if bits == 8 else torch.int8)
    torch.bitwise_and(weight, (2 ** bits) - 1, out=weight)
    weight = weight.reshape(-1, group_size, weight.shape[2])

    # Apply dequantization formula and reshape the final weight
    weight = (scales * (weight - zeros))
    weight = weight.reshape(weight.shape[0] * weight.shape[1], weight.shape[2])

    # Return the transposed weight
    return weight.transpose(0, 1)

# Function: Load quantized model and perform dequantization
def get_pytorch_bin():
    # Specify model file path
    path = "./your_model_folder/gptq_model-4bit-128g.bin"

    # Dictionary to store processed weights
    tensors = {}

    # Load the model file
    f = torch.load(path, map_location="cpu")

    # Iterate through keys in the model
    for idx, k in enumerate(f.keys()):
        ori_w = f[k]  # Original weight
        keys = k  # Original key name

        # Skip non-weight entries
        if ".qzeros" in k or ".scales" in k or ".g_idx" in k:
            continue
        if "o_proj.bias" in k or "up_proj.bias" in k or "down_proj.bias" in k or "gate_proj.bias" in k:
            continue

        # Process quantized weights
        if ".qweight" in k:
            qweight = f[k]  # Quantized weight
            qzeros = f[k.replace(".qweight", ".qzeros")]  # Zero points
            scales = f[k.replace(".qweight", ".scales")]  # Scales
            g_idx = f[k.replace(".qweight", ".g_idx")]  # Group index
            ori_w = dequantization(qweight, qzeros, scales, g_idx)  # Perform dequantization
            keys = k.replace(".qweight", ".weight")  # Update key name

        # Add processed weight to the dictionary
        tensors[keys] = ori_w

    # Print the number of processed weights and save as a new model file
    print(len(tensors))
    torch.save(tensors, "./your_model_folder/pytorch_model.bin")

# Main program entry point
if __name__ == '__main__':
    get_pytorch_bin()
相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
Password Pytorch 递归学习法 GIT Jupyter InvalidArgumentError 多进程 GPT4 ModelScope 净利润 Quantize Markdown 音频 Hilton CC Video 关于博主 PDF Tiktoken tqdm Card Disk 公式 强化学习 Attention XML Docker FP64 Datetime Heatmap LeetCode 云服务器 Agent Plate Sklearn Review DeepStream FP16 GGML SAM Bipartite Bert ONNX scipy DeepSeek Animate ChatGPT TensorFlow Ptyhon 第一性原理 FP8 域名 BF16 飞书 Logo SPIE Permission Linux Web Bin Translation RAR Github printf Diagram 图形思考法 Tensor TensorRT Pillow VGG-16 NLTK Shortcut HaggingFace Nginx C++ 版权 Search Django 证件照 LLM Transformers BeautifulSoup Base64 Hungarian Python CTC 算法题 v0.dev GPTQ torchinfo 顶会 EXCEL LoRA git Excel 论文 Breakpoint SVR Statistics 报税 llama.cpp IndexTTS2 Paper WAN YOLO COCO git-lfs uwsgi Algorithm icon Tracking NameSilo Streamlit 图标 VSCode RL GoogLeNet FP32 Conda Baidu OCR Paddle OpenCV Gemma uWSGI Safetensors FlashAttention Zip Pickle PDB logger Michelin Windows Qwen2 继承 搞笑 Template Hotel Crawler Firewall Interview Qwen2.5 Google Freesound ms-swift Augmentation Random Distillation UNIX mmap Knowledge HuggingFace TTS Data Cloudreve Bitcoin Domain VPN Miniforge NLP RGB CLAP CUDA OpenAI Math Image2Text PyCharm Jetson Mixtral SQLite Land tar Magnet Llama Dataset LaTeX WebCrawler 签证 Vmess 阿里云 transformers API FastAPI AI 多线程 PyTorch CAM CEIR QWEN Git v2ray Pandas Qwen CV Color News Numpy Food Use hf Proxy UI JSON Website 财报 BTC ResNet-50 PIP CSV diffusers Quantization XGBoost MD5 论文速读 Plotly SQL Clash 腾讯云 Anaconda Input TSV Ubuntu LLAMA Rebuttal Claude Vim
站点统计

本站现有博文333篇,共被浏览917077

本站已经建立2621天!

热门文章
文章归档
回到顶部