EADST

Pytorch Q4_1 Quantize and Dequantize aligning with llama.cpp

Pytorch Q4_1 Quantize and Dequantize aligning with llama.cpp

import torch

# Check if CUDA is available
use_cuda = torch.cuda.is_available()
device = torch.device("cuda" if use_cuda else "cpu")

def q4_1_quantize_and_dequantize_tensor(tensor):
    tensor = tensor.to(dtype=torch.float32, device=device)

    # Reshape tensor to process each 4-value block independently
    orig_shape = tensor.shape
    tensor = tensor.view(-1, 32)

    # Find the min and max values per block
    min_vals = torch.min(tensor, dim=1)[0]
    max_vals = torch.max(tensor, dim=1)[0]

    # Calculate scale d for each block
    d = (max_vals - min_vals) / (2**4 - 1)
    d[d == 0] = 1.0  # Prevent division by zero

    # Calculate inverse of d
    ids = 1.0 / d

    # Quantize tensor elements
    quantized_tensors = (tensor - min_vals[:, None]) * ids[:, None]

    # Clamp values to be between 0 and 15 (for 4 bits)
    quantized_tensors = torch.clamp(quantized_tensors + 0.5, 0, 15).to(torch.uint8)

    # Dequantize the tensor
    dequantized_tensors = (quantized_tensors.float() * d[:, None]) + min_vals[:, None]

    # Reshape back to the original shape
    dequantized_tensors = dequantized_tensors.view(orig_shape).to(dtype=torch.float16)

    return dequantized_tensors

# Assuming 'model_part' is already loaded and on CPU
model_part = torch.load(f"your_model_path/pytorch_model.bin", map_location="cpu")
keywords = [
    "embed_tokens.weight",
    "self_attn.q_proj.weight",
    "self_attn.k_proj.weight",
    "self_attn.v_proj.weight",
    "self_attn.o_proj.weight",
    "mlp.up_proj.weight",
    "mlp.gate_proj.weight",
    "mlp.down_proj.weight",
    "lm_head.weight"
]
for name, data in model_part.items():
    for word in keywords:
        if word in name:
            # Quantize and dequantize the entire tensor
            model_part[name] = q4_1_quantize_and_dequantize_tensor(data)

# Save the updated model parts
torch.save(model_part, "pytorch_model_quantized.bin")

Reference:

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
MD5 transformers Video FP16 顶会 域名 TensorFlow SQL LeetCode 继承 Template Markdown Tiktoken Claude 净利润 搞笑 Dataset Safetensors Interview 版权 Algorithm Land C++ Cloudreve Disk Transformers Github CAM Color Qwen2 ChatGPT BTC uWSGI 阿里云 CC InvalidArgumentError Augmentation 公式 printf 图形思考法 OpenAI CSV Review Use Michelin NLTK TTS Qwen Diagram SAM LoRA Mixtral GIT CTC FP32 API GGML LaTeX Translation Quantization Django Distillation Quantize Crawler tar ms-swift Gemma llama.cpp Conda ModelScope Pytorch Logo Card Windows VGG-16 Firewall Attention 论文 Pillow LLM 云服务器 FP8 SPIE 证件照 Anaconda LLAMA scipy CUDA Domain CV Hotel Llama PyCharm v0.dev Ptyhon Web Data 第一性原理 BeautifulSoup QWEN PyTorch GPT4 FP64 v2ray RAR hf Agent PIP Pandas WAN Miniforge FlashAttention Excel TSV UNIX 算法题 Random SQLite TensorRT Plate diffusers 关于博主 VPN YOLO 报税 Proxy 财报 Breakpoint icon 多进程 NameSilo CLAP COCO Vmess git Freesound DeepStream mmap OpenCV ResNet-50 UI Jetson SVR Clash Git Statistics Bipartite Food Knowledge ONNX RL Sklearn Linux Paddle Website Paper Hungarian Permission 签证 Nginx Tracking Rebuttal XGBoost 强化学习 RGB Numpy CEIR Bitcoin IndexTTS2 Jupyter Shortcut Search FastAPI Bert Python AI Input Bin Math Image2Text tqdm Vim OCR Datetime HaggingFace Password logger git-lfs Streamlit EXCEL Pickle 音频 腾讯云 Tensor GoogLeNet DeepSeek PDF 多线程 VSCode Heatmap 论文速读 Docker PDB JSON Google uwsgi 飞书 Magnet GPTQ Ubuntu Qwen2.5 torchinfo 图标 NLP BF16 递归学习法 Baidu WebCrawler Zip XML Animate Plotly Base64 News Hilton HuggingFace
站点统计

本站现有博文333篇,共被浏览922072

本站已经建立2628天!

热门文章
文章归档
回到顶部