EADST

Pytorch Q4_1 Quantize and Dequantize aligning with llama.cpp

Pytorch Q4_1 Quantize and Dequantize aligning with llama.cpp

import torch

# Check if CUDA is available
use_cuda = torch.cuda.is_available()
device = torch.device("cuda" if use_cuda else "cpu")

def q4_1_quantize_and_dequantize_tensor(tensor):
    tensor = tensor.to(dtype=torch.float32, device=device)

    # Reshape tensor to process each 4-value block independently
    orig_shape = tensor.shape
    tensor = tensor.view(-1, 32)

    # Find the min and max values per block
    min_vals = torch.min(tensor, dim=1)[0]
    max_vals = torch.max(tensor, dim=1)[0]

    # Calculate scale d for each block
    d = (max_vals - min_vals) / (2**4 - 1)
    d[d == 0] = 1.0  # Prevent division by zero

    # Calculate inverse of d
    ids = 1.0 / d

    # Quantize tensor elements
    quantized_tensors = (tensor - min_vals[:, None]) * ids[:, None]

    # Clamp values to be between 0 and 15 (for 4 bits)
    quantized_tensors = torch.clamp(quantized_tensors + 0.5, 0, 15).to(torch.uint8)

    # Dequantize the tensor
    dequantized_tensors = (quantized_tensors.float() * d[:, None]) + min_vals[:, None]

    # Reshape back to the original shape
    dequantized_tensors = dequantized_tensors.view(orig_shape).to(dtype=torch.float16)

    return dequantized_tensors

# Assuming 'model_part' is already loaded and on CPU
model_part = torch.load(f"your_model_path/pytorch_model.bin", map_location="cpu")
keywords = [
    "embed_tokens.weight",
    "self_attn.q_proj.weight",
    "self_attn.k_proj.weight",
    "self_attn.v_proj.weight",
    "self_attn.o_proj.weight",
    "mlp.up_proj.weight",
    "mlp.gate_proj.weight",
    "mlp.down_proj.weight",
    "lm_head.weight"
]
for name, data in model_part.items():
    for word in keywords:
        if word in name:
            # Quantize and dequantize the entire tensor
            model_part[name] = q4_1_quantize_and_dequantize_tensor(data)

# Save the updated model parts
torch.save(model_part, "pytorch_model_quantized.bin")

Reference:

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
Diagram CAM FP16 RL PyTorch CLAP Ptyhon Conda NameSilo hf XML uWSGI BTC WebCrawler 证件照 VGG-16 Clash Disk FP8 CTC FastAPI DeepSeek C++ Bitcoin Qwen2.5 logger 飞书 Mixtral Django Distillation SQL JSON Animate TSV Crawler Plotly BeautifulSoup Hilton uwsgi git-lfs transformers Llama Land Knowledge COCO mmap PyCharm Quantize Pickle LLM OCR HuggingFace OpenAI Use TTS TensorRT Jetson SPIE CC API Sklearn Math Cloudreve Tensor DeepStream EXCEL Ubuntu HaggingFace Miniforge 搞笑 Pillow 财报 IndexTTS2 Freesound OpenCV Food Translation Numpy ModelScope Paper UI Jev 多线程 Claude torchinfo VPN printf 音频 Google InvalidArgumentError Video TensorFlow Markdown Algorithm NLP CUDA LaTeX BF16 Pandas Input CSV 域名 Docker Augmentation GPT4 XGBoost Quantization 强化学习 多进程 报税 Magnet Random Website 云服务器 PIP Hungarian Michelin Shortcut LLAMA Windows Web tar diffusers RGB Firewall Qwen2 FP32 Bert SVR 算法题 版权 Gemma LeetCode Nginx Zip Pytorch ONNX PDB git Harness AI 关于博主 Vmess 图标 公式 llama.cpp Statistics PDF GoogLeNet 第一性原理 Search Safetensors 继承 Tiktoken VSCode tqdm Streamlit 图形思考法 QWEN NLTK UNIX Attention FP64 Transformers Vim Interview Baidu Rebuttal v2ray Data LoRA Github scipy ResNet-50 GPTQ v0.dev Linux 论文速读 SAM Plate 腾讯云 Password SQLite Image2Text API网关 Paddle Heatmap Proxy 顶会 Logo Excel Domain YOLO Bin Dataset 递归学习法 论文 Jupyter CV Datetime News RAR Card CEIR Bipartite WAN Tracking icon 阿里云 MD5 Anaconda Agent Breakpoint GIT Base64 ChatGPT FlashAttention Template 净利润 ms-swift Python Hotel Qwen Git GGML Review 签证 Permission Color
站点统计

本站现有博文337篇,共被浏览961756次

本站已经建立2673天!

热门文章
文章归档
回到顶部