[
    {
        "id": "osp-17794",
        "type": "article-journal",
        "title": "AdSpark: A Large-Scale Dataset and Benchmark for Product-Centric Advertisement Video Generation",
        "author": [
            {
                "family": "Yang",
                "given": "Zhifei"
            },
            {
                "family": "Jiang",
                "given": "Zhao"
            },
            {
                "family": "Lu",
                "given": "Keyang"
            },
            {
                "family": "Zhu",
                "given": "Honghe"
            },
            {
                "family": "Zhang",
                "given": "Zheng"
            },
            {
                "family": "Lv",
                "given": "Jingjing"
            },
            {
                "family": "Peng",
                "given": "Changping"
            },
            {
                "family": "Law",
                "given": "Ching"
            },
            {
                "family": "Xiao",
                "given": "Zhen"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/adspark-a-large-scale-dataset-and-benchmark-for-product-centric-advertisement-video-generation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Product-centric advertisement video generation aims to create promotional videos that preserve fine-grained product identity while presenting selling points through coherent multi-shot narratives. However, this emerging task remains underexplored due to the lack of large-scale advertisement-specific datasets and comprehensive evaluation frameworks. To address this gap, we introduce \\textbf{AdSpark}, a large-scale dataset and benchmark for product-centric advertisement video generation, based on data from a major e-commerce platform. \\textit{AdSpark-300K} contains approximately 300K reference image--prompt--video triplets, comprising a real-world subset and a synthetic subset. Each sample provides structured advertisement annotations, including product identity annotations, selling-point descriptions, creative plans, and aligned audio scripts, enabling models to learn product preservation and advertisement-oriented visual storytelling. We further propose \\textit{AdSpark-Bench}, a diagnostic benchmark that evaluates generated advertisements across six dimensions, including visual quality, product fidelity, instruction adherence, temporal coherence, audio alignment, and advertisement effectiveness. Based on AdSpark-Bench, we evaluate representative models, revealing key challenges in product preservation, multi-shot storytelling, and selling-point visualization. Experiments with AdSpark-300K-finetuned models further validate the effectiveness of our dataset. AdSpark provides a unified dataset and benchmark for future research, and we will release the dataset upon acceptance."
    }
]