[
    {
        "id": "osp-15422",
        "type": "article-journal",
        "title": "LayerRoPE: Dynamic Depth-wise Magnitude & Angular Superposition",
        "author": [
            {
                "family": "Srivastava",
                "given": "Shikhar"
            },
            {
                "family": "Kanan",
                "given": "Christopher"
            }
        ],
        "URL": "https://omanscience.com/en/articles/layerrope-dynamic-depth-wise-magnitude-angular-superposition",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "As data propagates through a Transformer, the norm of its hidden states grows by orders of magnitude with depth, a phenomenon framed as 'curse of depth' and nearly universally treated as a pathology to be suppressed. We take the opposite view. Across 16 pre-trained LLMs from 9 families, spanning dense, mixture-of-experts and hybrid architectures and Pre-, Peri- and Post-Norm designs, we find that this growth reflects an emergent depth-positional encoding, carried by the only learned per-layer gain on the residual stream, the normalization weight $γ$: with depth, $γ$ grows in magnitude and rotates in direction, jointly encoding the layer index. We make this depth-conditioned encoding explicit with LayerRoPE, an implicit analog of RoPE along the depth axis, which replaces all layerwise $γ$ vectors with a single shared vector and depth-conditioned scalars, at a net reduction in parameters and $"
    }
]