EADST

Pytorch Q4_0 Quantize and Dequantize aligning with llama.cpp

Pytorch Q4_0 Quantize and Dequantize aligning with llama.cpp

import torch

# Check if CUDA is available
use_cuda = torch.cuda.is_available()
device = torch.device("cuda" if use_cuda else "cpu")

def q4_0_quantize_and_dequantize_tensor(tensor):
    tensor = tensor.to(dtype=torch.float32, device=device)

    # Reshape tensor to process each 32-value block independently
    orig_shape = tensor.shape
    tensor = tensor.view(-1, 32)

    # Find the maximum absolute value per block
    max_vals = torch.max(torch.abs(tensor), dim=1)[0]

    # Prevent division by zero
    max_vals[max_vals == 0] = 1.0

    # Calculate d and id for each block
    d = max_vals / -8.0
    ids = 1.0 / d

    # Scale and quantize tensor elements
    scaled_tensors = tensor * ids[:, None]
    quantized_tensors = torch.clamp(scaled_tensors + 8.5, 0, 15).to(torch.uint8)

    # Dequantize the tensor
    dequantized_tensors = (quantized_tensors.float() - 8.0) * d[:, None]

    # Reshape back to the original shape
    dequantized_tensors = dequantized_tensors.view(orig_shape).to(dtype=torch.float16)

    return dequantized_tensors

# Assuming 'model_part' is already loaded and on CPU
model_part = torch.load(f"your_model_path/pytorch_model.bin", map_location="cpu")
keywords = [
    "embed_tokens.weight",
    "self_attn.q_proj.weight",
    "self_attn.k_proj.weight",
    "self_attn.v_proj.weight",
    "self_attn.o_proj.weight",
    "mlp.up_proj.weight",
    "mlp.gate_proj.weight",
    "mlp.down_proj.weight",
    "lm_head.weight"
]
for name, data in model_part.items():
    for word in keywords:
        if word in name:
            # Quantize and dequantize the entire tensor
            model_part[name] = q4_0_quantize_and_dequantize_tensor(data)

# Save the updated model parts
torch.save(model_part, "pytorch_model_quantized.bin")

Reference:

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
JSON NameSilo logger Base64 YOLO Paddle Nginx Llama transformers GGML Animate Hungarian FP16 QWEN Heatmap 财报 Paper 多线程 DeepSeek FlashAttention 飞书 Template Interview Tracking Numpy Zip VGG-16 FP8 Plotly BF16 Google Permission Random Linux Agent FP64 Python PyTorch CTC UI 公式 API HuggingFace Web Augmentation 顶会 OCR Tiktoken Qwen2.5 NLP Firewall mmap PDF GPT4 Color v0.dev CC 图标 Cloudreve 音频 Disk COCO RL CAM 算法题 Windows Magnet BTC XML Pandas Password WebCrawler Claude icon 多进程 Image2Text uwsgi InvalidArgumentError C++ 递归学习法 FP32 Jupyter Bitcoin 继承 Domain TensorRT NLTK GPTQ Distillation Breakpoint Video OpenCV RGB Qwen2 Algorithm Harness Vim v2ray 云服务器 LeetCode 第一性原理 强化学习 Website CSV TensorFlow LoRA hf SQL CUDA 阿里云 ResNet-50 Quantization Statistics XGBoost 论文速读 Anaconda Plate Datetime DeepStream Land SPIE Hotel PIP RAR Use Pickle Sklearn LLAMA diffusers Attention Mixtral torchinfo FastAPI Git Review EXCEL ChatGPT Freesound Michelin git LaTeX 搞笑 ONNX CEIR SVR SAM git-lfs CLAP Data Pillow API网关 Safetensors Card uWSGI Docker Knowledge WAN News Logo TTS Bin Input AI Jetson Miniforge Ubuntu VPN MD5 PyCharm Search Conda Tensor Crawler llama.cpp Food 关于博主 Hilton 版权 Clash Baidu Dataset Streamlit GoogLeNet Markdown Rebuttal LLM GIT TSV 图形思考法 Shortcut Github ModelScope Django Proxy VSCode Bert Transformers 论文 UNIX SQLite Ptyhon 净利润 Bipartite 签证 Vmess Translation Jev Diagram PDB Math CV Excel IndexTTS2 tar BeautifulSoup 腾讯云 OpenAI scipy ms-swift 域名 报税 printf tqdm Quantize HaggingFace 证件照 Qwen Gemma Pytorch
站点统计

本站现有博文337篇,共被浏览961407次

本站已经建立2673天!

热门文章
文章归档
回到顶部