EADST

Pytorch GPTQ Dequantizing Function

Pytorch GPTQ Dequantizing Function

Here is the Python code optimizing the dequantization of a GPTQ model to torch FP16 format.

import torch

# Function: Dequantize quantized weights
def dequantization(qweight, qzeros, scales, g_idx, bits=4, group_size=128, device='cuda:0'):
    # Create a tensor for bitwise right shift operation
    wf = torch.tensor(list(range(0, 32, bits)), dtype=torch.int32).unsqueeze(0)

    # Apply bitwise right shift and convert qzeros to the appropriate type
    zeros = torch.bitwise_right_shift(torch.unsqueeze(qzeros, 2).expand(-1, -1, 32 // bits), wf.unsqueeze(0)).to(torch.int16 if bits == 8 else torch.int8)
    torch.bitwise_and(zeros, (2 ** bits) - 1, out=zeros)

    # Reshape the zeros tensor
    zeros = zeros + 1
    zeros = zeros.reshape(-1, 1, zeros.shape[1] * zeros.shape[2])

    # Reshape the scales tensor
    scales = scales.reshape(-1, 1, scales.shape[-1])

    # Similar bitwise right shift operation for qweight and reshape
    weight = torch.bitwise_right_shift(torch.unsqueeze(qweight, 1).expand(-1, 32 // bits, -1), wf.unsqueeze(-1)).to(torch.int16 if bits == 8 else torch.int8)
    torch.bitwise_and(weight, (2 ** bits) - 1, out=weight)
    weight = weight.reshape(-1, group_size, weight.shape[2])

    # Apply dequantization formula and reshape the final weight
    weight = (scales * (weight - zeros))
    weight = weight.reshape(weight.shape[0] * weight.shape[1], weight.shape[2])

    # Return the transposed weight
    return weight.transpose(0, 1)

# Function: Load quantized model and perform dequantization
def get_pytorch_bin():
    # Specify model file path
    path = "./your_model_folder/gptq_model-4bit-128g.bin"

    # Dictionary to store processed weights
    tensors = {}

    # Load the model file
    f = torch.load(path, map_location="cpu")

    # Iterate through keys in the model
    for idx, k in enumerate(f.keys()):
        ori_w = f[k]  # Original weight
        keys = k  # Original key name

        # Skip non-weight entries
        if ".qzeros" in k or ".scales" in k or ".g_idx" in k:
            continue
        if "o_proj.bias" in k or "up_proj.bias" in k or "down_proj.bias" in k or "gate_proj.bias" in k:
            continue

        # Process quantized weights
        if ".qweight" in k:
            qweight = f[k]  # Quantized weight
            qzeros = f[k.replace(".qweight", ".qzeros")]  # Zero points
            scales = f[k.replace(".qweight", ".scales")]  # Scales
            g_idx = f[k.replace(".qweight", ".g_idx")]  # Group index
            ori_w = dequantization(qweight, qzeros, scales, g_idx)  # Perform dequantization
            keys = k.replace(".qweight", ".weight")  # Update key name

        # Add processed weight to the dictionary
        tensors[keys] = ori_w

    # Print the number of processed weights and save as a new model file
    print(len(tensors))
    torch.save(tensors, "./your_model_folder/pytorch_model.bin")

# Main program entry point
if __name__ == '__main__':
    get_pytorch_bin()
相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
CLAP Quantization Permission Rebuttal GGML Input Template MD5 论文 Agent WebCrawler 图标 Hungarian Github AI GoogLeNet Tiktoken 报税 Heatmap Safetensors 图形思考法 OpenCV tqdm CV Use WAN git Numpy Jetson Tensor Conda Streamlit Qwen2 Attention Hotel NLTK Ubuntu scipy SAM Data TensorRT HaggingFace InvalidArgumentError Plotly Vim llama.cpp OpenAI hf 论文速读 Crawler Logo PyTorch uWSGI 飞书 UNIX Paddle 版权 Anaconda VSCode Cloudreve DeepSeek PyCharm 阿里云 PDB JSON v2ray 递归学习法 Llama PIP 搞笑 printf Pytorch LaTeX Knowledge Pickle VPN XGBoost GIT FlashAttention RL DeepStream Datetime TTS Jupyter Color Augmentation Nginx Tracking 多进程 ModelScope Django Animate Review Michelin API 公式 Dataset BF16 Qwen SQLite Translation BTC Interview Algorithm C++ Miniforge Bin Python Password Image2Text Domain Markdown FP16 TSV RAR 财报 BeautifulSoup 第一性原理 NLP Paper News 强化学习 UI Windows Gemma 多线程 Google RGB Food 证件照 音频 腾讯云 Baidu Video LeetCode Transformers Ptyhon COCO ms-swift Card GPTQ NameSilo EXCEL Docker Sklearn 算法题 FP64 Shortcut 继承 ChatGPT GPT4 签证 uwsgi Pandas Magnet tar Freesound IndexTTS2 TensorFlow QWEN Random v0.dev YOLO Bitcoin Excel 域名 Zip logger CEIR Disk Distillation LoRA SPIE Git SVR Proxy torchinfo icon 云服务器 ONNX CTC OCR Website Land FastAPI Breakpoint CSV 净利润 XML LLAMA PDF Claude Bipartite LLM Linux Search Clash Quantize diffusers FP8 Hilton 关于博主 顶会 Vmess CUDA Diagram Bert SQL VGG-16 ResNet-50 Base64 Pillow transformers mmap Qwen2.5 Math HuggingFace Statistics Web Mixtral git-lfs CAM CC Plate Firewall FP32
站点统计

本站现有博文333篇,共被浏览922337

本站已经建立2628天!

热门文章
文章归档
回到顶部