[
    {
        "id": "osp-23033",
        "type": "article-journal",
        "title": "G$^2$PTQ: Improving LLM Post-Training Quantization with Generalized Gradient Compensation",
        "author": [
            {
                "family": "Liu",
                "given": "Ruikang"
            },
            {
                "family": "Bai",
                "given": "Haoli"
            },
            {
                "family": "Sun",
                "given": "Yuxuan"
            },
            {
                "family": "Zhang",
                "given": "Qian"
            },
            {
                "family": "Cai",
                "given": "Wenzheng"
            },
            {
                "family": "Hao",
                "given": "Yanqi"
            },
            {
                "family": "Wang",
                "given": "Feiyu"
            },
            {
                "family": "Zhong",
                "given": "Weidong"
            },
            {
                "family": "Wang",
                "given": "Zhuang"
            },
            {
                "family": "Yang",
                "given": "Tong"
            },
            {
                "family": "Zhou",
                "given": "Xiangsheng"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/g-2-ptq-improving-llm-post-training-quantization-with-generalized-gradient-compensation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Post-training quantization (PTQ) is a practical approach to reducing the memory and computational footprint of large language models (LLMs) without retraining. GPTQ-based methods have become the de facto standard, yet they suffer from two complementary limitations. Methods with local, layer-wise objectives lack global supervision; while methods with global objectives fix their Hessian estimates at the start and ignore first-order gradients, so their guidance grows stale as quantization proceeds. This paper presents G$^2$PTQ, a unified PTQ framework with Generalized Gradient Compensation that integrates both first- and second-order information under a globally supervised, block-wise optimization objective. By refreshing gradient and Hessian estimates before quantizing each Transformer block, G$^2$PTQ avoids the staleness of prior global methods. Furthermore, to stabilize the exact first-order compensation, we introduce a trust-region scaling mechanism that dynamically bounds the gradient step to prevent exploding weight updates. Finally, we derive efficient implementations for block-wise Hessian approximation and exact gradient compensation. Experimental results on various model families and bit-widths demonstrate that G$^2$PTQ enables better alignment with the full-precision model, outperforming state-of-the-art baselines. Code is available at: https://github.com/G2PTQ/G2PTQ."
    }
]