[
    {
        "id": "osp-20891",
        "type": "article-journal",
        "title": "Telescopic Language Models",
        "author": [
            {
                "family": "Guo",
                "given": "Zhilin"
            },
            {
                "family": "Zhang",
                "given": "Boqiao"
            },
            {
                "family": "Aktas",
                "given": "Hakan"
            },
            {
                "family": "Fogarty",
                "given": "Kyle"
            },
            {
                "family": "Aslan",
                "given": "Nursena Koprucu"
            },
            {
                "family": "Li",
                "given": "Wenzhao"
            },
            {
                "family": "Baykal",
                "given": "Canberk"
            },
            {
                "family": "Miao",
                "given": "Albert"
            },
            {
                "family": "Hong",
                "given": "Siyu"
            },
            {
                "family": "Liu",
                "given": "Yixiao"
            },
            {
                "family": "Wu",
                "given": "Adam"
            },
            {
                "family": "Singh",
                "given": "Ashish Kumar"
            },
            {
                "family": "Khattar",
                "given": "Sakar"
            },
            {
                "family": "Zhou",
                "given": "Chenliang"
            },
            {
                "family": "Xia",
                "given": "Weihao"
            },
            {
                "family": "Vasconcelos",
                "given": "Cristina Nader"
            },
            {
                "family": "Oztireli",
                "given": "Cengiz"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/telescopic-language-models",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "One deployed language model must often serve many compute budgets, yet serving each budget still means a separate training or compression run per point. We train a Telescopic Language Model (TLM) to be that continuum: a nested-capacity Transformer supervised by stochastic prefix supervision with a full anchor. At every step, one randomly truncated prefix of the capacity axis is trained against the full next-token target, alongside one full-capacity pass, so the trained artifact is a valid language model at every depth. Two forward-backward passes per step, no architectural change, nothing extra at inference. Fixed-exit suites such as Matryoshka Language Model Suites (MLMS) occupy one point in this design space, and the point has a cost: supervising only a few fixed exits leaves the nested model at chance level everywhere else (perplexity 10^2-10^5 in our baselines). On a 200M proxy suite (20B FineWeb-Edu tokens, identical data stream for all methods), a single TLM run is a valid language model at every one of its twenty layer prefixes, in perplexity and on perplexity-sensitive downstream tasks, reducing the area under the quality-budget curve by 43-44% relative to the fixed-exit suites while matching them at full capacity, at ~12% lower GPU cost per run. The prefix sampling density is a dial: concentrating it on a few depths recovers fixed-exit quality there at the price of the continuum, so the operating points become a training-time choice rather than an architectural one. These results indicate that the training objective, not the nesting itself, is what makes a model elastic."
    }
]