EADST

Pytorch GPTQ Dequantizing Function

Pytorch GPTQ Dequantizing Function

Here is the Python code optimizing the dequantization of a GPTQ model to torch FP16 format.

import torch

# Function: Dequantize quantized weights
def dequantization(qweight, qzeros, scales, g_idx, bits=4, group_size=128, device='cuda:0'):
    # Create a tensor for bitwise right shift operation
    wf = torch.tensor(list(range(0, 32, bits)), dtype=torch.int32).unsqueeze(0)

    # Apply bitwise right shift and convert qzeros to the appropriate type
    zeros = torch.bitwise_right_shift(torch.unsqueeze(qzeros, 2).expand(-1, -1, 32 // bits), wf.unsqueeze(0)).to(torch.int16 if bits == 8 else torch.int8)
    torch.bitwise_and(zeros, (2 ** bits) - 1, out=zeros)

    # Reshape the zeros tensor
    zeros = zeros + 1
    zeros = zeros.reshape(-1, 1, zeros.shape[1] * zeros.shape[2])

    # Reshape the scales tensor
    scales = scales.reshape(-1, 1, scales.shape[-1])

    # Similar bitwise right shift operation for qweight and reshape
    weight = torch.bitwise_right_shift(torch.unsqueeze(qweight, 1).expand(-1, 32 // bits, -1), wf.unsqueeze(-1)).to(torch.int16 if bits == 8 else torch.int8)
    torch.bitwise_and(weight, (2 ** bits) - 1, out=weight)
    weight = weight.reshape(-1, group_size, weight.shape[2])

    # Apply dequantization formula and reshape the final weight
    weight = (scales * (weight - zeros))
    weight = weight.reshape(weight.shape[0] * weight.shape[1], weight.shape[2])

    # Return the transposed weight
    return weight.transpose(0, 1)

# Function: Load quantized model and perform dequantization
def get_pytorch_bin():
    # Specify model file path
    path = "./your_model_folder/gptq_model-4bit-128g.bin"

    # Dictionary to store processed weights
    tensors = {}

    # Load the model file
    f = torch.load(path, map_location="cpu")

    # Iterate through keys in the model
    for idx, k in enumerate(f.keys()):
        ori_w = f[k]  # Original weight
        keys = k  # Original key name

        # Skip non-weight entries
        if ".qzeros" in k or ".scales" in k or ".g_idx" in k:
            continue
        if "o_proj.bias" in k or "up_proj.bias" in k or "down_proj.bias" in k or "gate_proj.bias" in k:
            continue

        # Process quantized weights
        if ".qweight" in k:
            qweight = f[k]  # Quantized weight
            qzeros = f[k.replace(".qweight", ".qzeros")]  # Zero points
            scales = f[k.replace(".qweight", ".scales")]  # Scales
            g_idx = f[k.replace(".qweight", ".g_idx")]  # Group index
            ori_w = dequantization(qweight, qzeros, scales, g_idx)  # Perform dequantization
            keys = k.replace(".qweight", ".weight")  # Update key name

        # Add processed weight to the dictionary
        tensors[keys] = ori_w

    # Print the number of processed weights and save as a new model file
    print(len(tensors))
    torch.save(tensors, "./your_model_folder/pytorch_model.bin")

# Main program entry point
if __name__ == '__main__':
    get_pytorch_bin()
相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
RL FP32 论文速读 OpenCV Conda AI Distillation SAM 继承 Hotel PyTorch 图标 Paper TensorRT Zip Safetensors 财报 VGG-16 llama.cpp CAM Streamlit ONNX Image2Text PIP Quantize Gemma 搞笑 飞书 C++ CLAP CV Pillow SQL 图形思考法 GIT v2ray 关于博主 InvalidArgumentError TensorFlow Michelin COCO Land SQLite TSV Anaconda Git 证件照 Magnet API 版权 ChatGPT WAN logger git GPTQ PyCharm Rebuttal News PDB MD5 Logo 报税 Hungarian Datetime Clash JSON Video Tensor DeepSeek 阿里云 ms-swift QWEN 腾讯云 Plate OpenAI Data Nginx 顶会 Bipartite Docker GGML 签证 Harness 净利润 ModelScope FP64 ResNet-50 NameSilo Baidu Use mmap TTS 递归学习法 音频 Diagram Pytorch IndexTTS2 Freesound LoRA 强化学习 hf uwsgi Python Plotly CEIR 论文 LeetCode YOLO Cloudreve Tiktoken PDF Quantization RGB tar scipy Color Random HuggingFace FlashAttention 多线程 API网关 GoogLeNet Excel WebCrawler GPT4 Password Pandas Translation Statistics transformers SPIE Qwen CTC Google CUDA Crawler Knowledge Sklearn VPN RAR 域名 LLAMA Domain Paddle Augmentation Claude VSCode 云服务器 第一性原理 uWSGI XGBoost Ptyhon Attention Agent DeepStream Jetson HaggingFace Vmess Search BeautifulSoup Qwen2.5 icon Vim CC Base64 公式 Proxy Animate Heatmap FP8 Web Permission Linux Review UNIX Qwen2 EXCEL Math Github SVR Django torchinfo UI BF16 Breakpoint NLP Ubuntu Food Llama Disk Interview Algorithm tqdm Jupyter Jev Firewall Tracking Windows v0.dev Dataset BTC Transformers printf Bin Mixtral FastAPI 算法题 Template CSV FP16 Pickle git-lfs LLM Numpy Shortcut NLTK Website LaTeX Bitcoin Input diffusers XML Miniforge 多进程 Hilton Markdown Bert Card OCR
站点统计

本站现有博文337篇,共被浏览960960次

本站已经建立2672天!

热门文章
文章归档
回到顶部