Copyright © 2026 Authors retain the copyright of this article. This article is an open access article distributed under the Creative Commons Attribution License which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.
@article{200524,
author = {Telukuntla Vinaykoushik and Ramabathin yashaswini and V Mohanapriya and MK Gowtham and Asis Marceline},
title = {AutoQuant: Automatic Mixed-Precision Quantization of LLMs under User-Specified Memory Budgets},
journal = {International Journal of Innovative Research in Technology},
year = {2026},
volume = {12},
number = {12},
pages = {1339-1345},
issn = {2349-6002},
url = {https://ijirt.org/article?manuscript=200524},
abstract = {Uniform-precision quantization forces a losing trade-off between compression depth and generation quality — AutoQuant eliminates it. AutoQuant calculates a composite per-layer sensitivity score from 99th-percentile-clipped squared weight gradients fused with activation magnitudes. It then drives a two-phase greedy allocator over (INT4, INT8, INT16) — locking insensitive layers at INT4 via a cliff heuristic and blocking unnecessary INT16 upgrades via a gated threshold — all without fine-tuning or labeled data. Quantized layers use weight-only symmetric quantization with grouped column scales (group size 128) and a per-output-row FP16 residual to make up for rounding errors. They work with Hugging Face causal LMs through a REST API and an interactive dashboard. Evaluated on GPT-2 and open Llama-family checkpoints, AutoQuant achieves up to 3.8× compression with less than 2% perplexity degradation relative to the FP16 reference — at identical compressed sizes, consistently outperforming uniform INT8 baselines.},
keywords = {Large Language Models, Mixed-Precision Quantization, Post-Training Quantization, Sensitivity Analysis, Model Compression, Weight-Only Quantization.},
month = {May},
}
Submit your research paper and those of your network (friends, colleagues, or peers) through your IPN account, and receive 800 INR for each paper that gets published.
Join NowNational Conference on Sustainable Engineering and Management - 2024 Last Date: 15th March 2024
Submit inquiry