EADST

Pytorch Q4_0 Quantize and Dequantize aligning with llama.cpp

Pytorch Q4_0 Quantize and Dequantize aligning with llama.cpp

import torch

# Check if CUDA is available
use_cuda = torch.cuda.is_available()
device = torch.device("cuda" if use_cuda else "cpu")

def q4_0_quantize_and_dequantize_tensor(tensor):
    tensor = tensor.to(dtype=torch.float32, device=device)

    # Reshape tensor to process each 32-value block independently
    orig_shape = tensor.shape
    tensor = tensor.view(-1, 32)

    # Find the maximum absolute value per block
    max_vals = torch.max(torch.abs(tensor), dim=1)[0]

    # Prevent division by zero
    max_vals[max_vals == 0] = 1.0

    # Calculate d and id for each block
    d = max_vals / -8.0
    ids = 1.0 / d

    # Scale and quantize tensor elements
    scaled_tensors = tensor * ids[:, None]
    quantized_tensors = torch.clamp(scaled_tensors + 8.5, 0, 15).to(torch.uint8)

    # Dequantize the tensor
    dequantized_tensors = (quantized_tensors.float() - 8.0) * d[:, None]

    # Reshape back to the original shape
    dequantized_tensors = dequantized_tensors.view(orig_shape).to(dtype=torch.float16)

    return dequantized_tensors

# Assuming 'model_part' is already loaded and on CPU
model_part = torch.load(f"your_model_path/pytorch_model.bin", map_location="cpu")
keywords = [
    "embed_tokens.weight",
    "self_attn.q_proj.weight",
    "self_attn.k_proj.weight",
    "self_attn.v_proj.weight",
    "self_attn.o_proj.weight",
    "mlp.up_proj.weight",
    "mlp.gate_proj.weight",
    "mlp.down_proj.weight",
    "lm_head.weight"
]
for name, data in model_part.items():
    for word in keywords:
        if word in name:
            # Quantize and dequantize the entire tensor
            model_part[name] = q4_0_quantize_and_dequantize_tensor(data)

# Save the updated model parts
torch.save(model_part, "pytorch_model_quantized.bin")

Reference:

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
VSCode Data SQLite RL COCO Pandas ms-swift Land Logo CTC logger Zip NLP SQL 报税 CC LaTeX uWSGI EXCEL Cloudreve 域名 tqdm OCR HaggingFace TSV 图标 飞书 Quantize FastAPI 签证 Dataset Pytorch PDF GIT Food 财报 Freesound FP8 AI NLTK Magnet tar Sklearn git-lfs Hilton 搞笑 净利润 Shortcut BTC Ubuntu BeautifulSoup hf Vim 腾讯云 Password uwsgi icon LLAMA Bipartite Proxy Crawler UNIX Gemma Clash CSV 关于博主 Tracking Conda Git 版权 HuggingFace API 第一性原理 Qwen Vmess 强化学习 LoRA 公式 ONNX Numpy Bitcoin Plotly v0.dev XGBoost Qwen2 Nginx Statistics Github Web CLAP 证件照 云服务器 Use 多进程 WAN Color Paper 音频 GoogLeNet CEIR Domain Permission PyTorch Mixtral RGB GGML Firewall Tensor Google OpenCV Agent Template FP32 Windows llama.cpp transformers Input Markdown scipy OpenAI Michelin LLM BF16 Hotel TensorFlow JSON IndexTTS2 Safetensors LeetCode Streamlit PDB SPIE Animate Distillation InvalidArgumentError Pickle Paddle Review Qwen2.5 Jupyter git 继承 FP64 ResNet-50 Website News Augmentation QWEN Llama 递归学习法 Tiktoken PyCharm FP16 Python mmap CAM Bin FlashAttention Translation Disk Base64 SAM Excel v2ray VPN Image2Text 论文速读 VGG-16 C++ XML Card DeepStream Video TTS YOLO GPT4 阿里云 Knowledge 图形思考法 Baidu Anaconda TensorRT NameSilo 顶会 多线程 Django WebCrawler Diagram SVR Bert GPTQ printf 算法题 DeepSeek Docker ModelScope Math ChatGPT Jetson Hungarian Pillow Miniforge Random Interview Heatmap MD5 UI Linux Search PIP Breakpoint torchinfo Attention Claude CV Ptyhon Rebuttal 论文 Transformers diffusers Quantization CUDA RAR Datetime Plate Algorithm
站点统计

本站现有博文333篇,共被浏览922224

本站已经建立2628天!

热门文章
文章归档
回到顶部