{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WV23MRAKMFBGEAXVZ6JONP6E37","short_pith_number":"pith:WV23MRAK","schema_version":"1.0","canonical_sha256":"b575b6440a61426202f5cf92e6bfc4dfc1b83ab8e97d382d09046fafe37bcb24","source":{"kind":"arxiv","id":"2206.03126","version":1},"attestation_state":"computed","paper":{"title":"Signal Propagation in Transformers: Theoretical Perspectives and the Role of Rank Collapse","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Antonio Orvieto, Aurelien Lucchi, Lorenzo Noci, Luca Biggio, Sidak Pal Singh, Sotiris Anagnostidis","submitted_at":"2022-06-07T09:07:24Z","abstract_excerpt":"Transformers have achieved remarkable success in several domains, ranging from natural language processing to computer vision. Nevertheless, it has been recently shown that stacking self-attention layers - the distinctive architectural component of Transformers - can result in rank collapse of the tokens' representations at initialization. The question of if and how rank collapse affects training is still largely unanswered, and its investigation is necessary for a more comprehensive understanding of this architecture. In this work, we shed new light on the causes and the effects of this pheno"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.03126","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-07T09:07:24Z","cross_cats_sorted":[],"title_canon_sha256":"6cdeb7aabb272cf3abc49a57708d97a355f81aac94ace690191eaa223f43a6db","abstract_canon_sha256":"cf87d8c3d4d1f5e739795a95fd45e8b8ab35984e2c81ece3f9b0046890e9cf47"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:29:39.586174Z","signature_b64":"67WxA1KZerJ4WfyOY0xPdbKLp31C7h/ucrbkhC1JOnqP3S21o6fL/Cwe3Cu+Ars0dejk4uJ/UxCwsZYMh6C6DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b575b6440a61426202f5cf92e6bfc4dfc1b83ab8e97d382d09046fafe37bcb24","last_reissued_at":"2026-07-05T04:29:39.585737Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:29:39.585737Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Signal Propagation in Transformers: Theoretical Perspectives and the Role of Rank Collapse","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Antonio Orvieto, Aurelien Lucchi, Lorenzo Noci, Luca Biggio, Sidak Pal Singh, Sotiris Anagnostidis","submitted_at":"2022-06-07T09:07:24Z","abstract_excerpt":"Transformers have achieved remarkable success in several domains, ranging from natural language processing to computer vision. Nevertheless, it has been recently shown that stacking self-attention layers - the distinctive architectural component of Transformers - can result in rank collapse of the tokens' representations at initialization. The question of if and how rank collapse affects training is still largely unanswered, and its investigation is necessary for a more comprehensive understanding of this architecture. In this work, we shed new light on the causes and the effects of this pheno"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.03126","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.03126/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.03126","created_at":"2026-07-05T04:29:39.585799+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.03126v1","created_at":"2026-07-05T04:29:39.585799+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.03126","created_at":"2026-07-05T04:29:39.585799+00:00"},{"alias_kind":"pith_short_12","alias_value":"WV23MRAKMFBG","created_at":"2026-07-05T04:29:39.585799+00:00"},{"alias_kind":"pith_short_16","alias_value":"WV23MRAKMFBGEAXV","created_at":"2026-07-05T04:29:39.585799+00:00"},{"alias_kind":"pith_short_8","alias_value":"WV23MRAK","created_at":"2026-07-05T04:29:39.585799+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21158","citing_title":"Dead-Direction Signatures: A Cheap Spectral Reading of Singular Complexity","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19491","citing_title":"Algebraic Dead Directions in LayerNorm Transformers: A Forward-Pass-Only Diagnostic at LLM Scale","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05957","citing_title":"Dead Directions: Geometric Singular Learning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2505.24333","citing_title":"Two failure modes of deep transformers and how to avoid them: a unified theory of signal propagation at initialisation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12697","citing_title":"A Unified Framework for Critical Scaling of Inverse Temperature in Self-Attention","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07925","citing_title":"Sinkhorn doubly stochastic attention rank decay analysis","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37","json":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37.json","graph_json":"https://pith.science/api/pith-number/WV23MRAKMFBGEAXVZ6JONP6E37/graph.json","events_json":"https://pith.science/api/pith-number/WV23MRAKMFBGEAXVZ6JONP6E37/events.json","paper":"https://pith.science/paper/WV23MRAK"},"agent_actions":{"view_html":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37","download_json":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37.json","view_paper":"https://pith.science/paper/WV23MRAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.03126&json=true","fetch_graph":"https://pith.science/api/pith-number/WV23MRAKMFBGEAXVZ6JONP6E37/graph.json","fetch_events":"https://pith.science/api/pith-number/WV23MRAKMFBGEAXVZ6JONP6E37/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37/action/storage_attestation","attest_author":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37/action/author_attestation","sign_citation":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37/action/citation_signature","submit_replication":"https://pith.science/pith/WV23MRAKMFBGEAXVZ6JONP6E37/action/replication_record"}},"created_at":"2026-07-05T04:29:39.585799+00:00","updated_at":"2026-07-05T04:29:39.585799+00:00"}