[
    {
        "id": "osp-15880",
        "type": "article-journal",
        "title": "GAMBIT: Learning to Plan Continuous Multi-Robot Trajectories",
        "author": [
            {
                "family": "Jain",
                "given": "Rishabh"
            },
            {
                "family": "Moldagalieva",
                "given": "Akmaral"
            },
            {
                "family": "Magnino",
                "given": "Lorenzo"
            },
            {
                "family": "Amir",
                "given": "Michael"
            },
            {
                "family": "Okumura",
                "given": "Keisuke"
            },
            {
                "family": "Shankar",
                "given": "Ajay"
            },
            {
                "family": "Hönig",
                "given": "Wolfgang"
            },
            {
                "family": "Prorok",
                "given": "Amanda"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/gambit-learning-to-plan-continuous-multi-robot-trajectories",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "GAMBIT is an opening chess move in which a player sacrifices a piece, typically a pawn, to gain a positional advantage later in the game. Analogously, in multi-robot coordination, individual robots may need to forgo locally reward-maximising behaviours to improve overall team performance. Such self-sacrificial behaviours are difficult to capture with manually designed heuristics, particularly in dense, interaction-rich environments. Focusing on double-integrator continuous dynamics, this work studies how to learn such coordinated heuristics over motion primitives for multi-robot trajectory execution. Our framework, GAMBIT, first learns coordinated motion-primitive selection through imitation learning and subsequently fine-tunes the policy through reinforcement learning. We further introduce a safeguarded rollout mechanism with backup trajectories that guarantees collision-free execution at all times. Experiments demonstrate that GAMBIT substantially outperforms a range of baselines, including centralised motion planners and decentralised reactive planners, while exhibiting strong scalability. In particular, it coordinates over a thousand robots with planning latency below a few hundred milliseconds in continuous domains."
    }
]