[
    {
        "id": "osp-15602",
        "type": "article-journal",
        "title": "TICDA: Tabular In-Context Data Attribution",
        "author": [
            {
                "family": "Benihaddadene",
                "given": "Yacine"
            },
            {
                "family": "Bhan",
                "given": "Milan"
            },
            {
                "family": "Dugelay",
                "given": "Eliot"
            },
            {
                "family": "Jawhar",
                "given": "Mohammed"
            },
            {
                "family": "Wong",
                "given": "Benjamin"
            },
            {
                "family": "Chesneau",
                "given": "Nicolas"
            },
            {
                "family": "Nguyen",
                "given": "Duong"
            }
        ],
        "URL": "https://omanscience.com/en/articles/ticda-tabular-in-context-data-attribution",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Tabular foundation models (TFMs) achieve strong predictive performance by conditioning on labeled demonstrations provided in context, without any parameter update. Yet how individual demonstrations shape a given prediction remains poorly understood. This gap matters in practice: the context is often assembled from whatever labeled data is available, potentially leading to the inclusion of mislabeled, redundant, or low-quality examples that degrade performance. Standard data attribution methods do not transfer to the TFM setting: resampling-based approaches such as DemoShapley require a combinatorial number of forward passes, and gradient-based estimators such as influence functions require computing training point's effect on the model parameters, which in-context learning never updates. We introduce TICDA, a method that measures the influence of every demonstration in the context directly from linear surrogates trained on TFM latent embeddings, in a single forward pass and at negligible cost. We show that TICDA offers the best compromise against competitors across four tasks: detecting labeling errors, curating context to preserve predictive accuracy while lowering inference cost, producing attribution scores that transfer across TFMs, and supporting an acquisition strategy for efficient active learning."
    }
]