[
    {
        "id": "osp-22266",
        "type": "article-journal",
        "title": "Alignment-Guided Flow Transformer for Efficient Vision-Language-Action Policy Learning",
        "author": [
            {
                "family": "Hu",
                "given": "Shengchao"
            },
            {
                "family": "Wang",
                "given": "Peng"
            },
            {
                "family": "Zhou",
                "given": "Qiyang"
            },
            {
                "family": "Zheng",
                "given": "Guodong"
            },
            {
                "family": "Huang",
                "given": "Yuqi"
            },
            {
                "family": "Shen",
                "given": "Li"
            },
            {
                "family": "Zhang",
                "given": "Ya"
            },
            {
                "family": "Tao",
                "given": "Dacheng"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/alignment-guided-flow-transformer-for-efficient-vision-language-action-policy-learning",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Recent advances in Vision-Language-Action (VLA) models point toward general-purpose robotic intelligence by unifying perception, instruction, and control. Despite impressive progress, existing VLA models often adapt poorly due to \\emph{tri-modal misalignment} among vision, language, and action, which weakens action grounding and hurts generalization and fine-tuning efficiency. In this work, we present Alignment-Guided Flow Transformer (AGFT), a novel framework that explicitly enforces tri-modal alignment through a dedicated alignment loss, bridging the representational gap across modalities and enhancing task adaptation. While prior research has predominantly emphasized bi-modal vision--language alignment, we systematically formalize and study tri-modal alignment in VLA models, and provide both ablations and analysis to isolate its role in improving adaptation and robustness. To further accelerate deployment, we adopt a flow-matching objective, enabling substantially fewer inference steps than diffusion-based policies while maintaining accuracy. Theoretically, we establish a quantitative connection between the tri-modal alignment gap and the optimization tightness of flow matching; empirically, experiments on the extensive benchmark show that AGFT achieves superior success rates and lower inference latency compared to SOTA baselines, underscoring tri-modal alignment as a key ingredient for scaling robust VLA manipulation."
    }
]