[
    {
        "id": "osp-15917",
        "type": "article-journal",
        "title": "Boosting Transferable Adversarial Attacks against Deep Reinforcement Learning",
        "author": [
            {
                "family": "Li",
                "given": "Zexin"
            },
            {
                "family": "Yao",
                "given": "Ruili"
            },
            {
                "family": "Zeng",
                "given": "Yiming"
            },
            {
                "family": "Gao",
                "given": "Xiaoxue"
            }
        ],
        "URL": "https://omanscience.com/en/articles/boosting-transferable-adversarial-attacks-against-deep-reinforcement-learning",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Most adversarial attacks on deep reinforcement learning (DRL) assume white-box access to the victim policy, which rarely holds in practice. This paper studies transfer-based black-box attacks on DRL: the attacker crafts observation perturbations on a white-box surrogate agent and feeds them to an unknown victim. We formulate the attack as return minimization under a per-step perturbation budget. We first show that transplanting transferable image-classification attacks (FGSM, MI-FGSM, and NI-FGSM) with a per-step objective yields perturbations that transfer but are no stronger than random noise of the same budget. We then propose a trajectory-level attack that optimizes a sequence of perturbations over a receding horizon through a differentiable model of the environment and a temperature-smoothed surrogate policy, with the same optimizers. On CartPole-v1 with ten DQN and DDQN agents and 100 surrogate--victim pairs, the trajectory-level attack outperforms per-step attacks and random noise in the white-box, cross-model, and cross-algorithm settings."
    }
]