{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:MXKNR2CR3XSBCG6OFT7UT5NJ6C","short_pith_number":"pith:MXKNR2CR","schema_version":"1.0","canonical_sha256":"65d4d8e851dde4111bce2cff49f5a9f0b6d22c3fec4a3ff74ed396e18458ab76","source":{"kind":"arxiv","id":"1908.04211","version":4},"attestation_state":"computed","paper":{"title":"On Identifiability in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dami\\'an Pascual, Gino Brunner, Massimiliano Ciaramita, Oliver Richter, Roger Wattenhofer, Yang Liu","submitted_at":"2019-08-12T15:48:34Z","abstract_excerpt":"In this paper we delve deep in the Transformer architecture by investigating two of its core components: self-attention and contextual embeddings. In particular, we study the identifiability of attention weights and token embeddings, and the aggregation of context into hidden tokens. We show that, for sequences longer than the attention head dimension, attention weights are not identifiable. We propose effective attention as a complementary tool for improving explanatory interpretations based on attention. Furthermore, we show that input tokens retain to a large degree their identity across th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.04211","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-08-12T15:48:34Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9a945acef436df0de8b6903abf257a616bc0c29cbe9820e74a8872f72ed27b94","abstract_canon_sha256":"c7f06dfeb71e4c1c5def093993ff39ca5b60a1dd5af6369f473a519d911ff793"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:39:05.393433Z","signature_b64":"ydalagS4KJEabS+DrmSsTzYt/sGFxLpI7G6+C/nDuUzMBU9j/nF3YvAvzEJUyc4ld5g8JjZ9o6RXNXaPDQJtDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65d4d8e851dde4111bce2cff49f5a9f0b6d22c3fec4a3ff74ed396e18458ab76","last_reissued_at":"2026-07-05T00:39:05.392944Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:39:05.392944Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Identifiability in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Dami\\'an Pascual, Gino Brunner, Massimiliano Ciaramita, Oliver Richter, Roger Wattenhofer, Yang Liu","submitted_at":"2019-08-12T15:48:34Z","abstract_excerpt":"In this paper we delve deep in the Transformer architecture by investigating two of its core components: self-attention and contextual embeddings. In particular, we study the identifiability of attention weights and token embeddings, and the aggregation of context into hidden tokens. We show that, for sequences longer than the attention head dimension, attention weights are not identifiable. We propose effective attention as a complementary tool for improving explanatory interpretations based on attention. Furthermore, we show that input tokens retain to a large degree their identity across th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.04211","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.04211/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.04211","created_at":"2026-07-05T00:39:05.392998+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.04211v4","created_at":"2026-07-05T00:39:05.392998+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.04211","created_at":"2026-07-05T00:39:05.392998+00:00"},{"alias_kind":"pith_short_12","alias_value":"MXKNR2CR3XSB","created_at":"2026-07-05T00:39:05.392998+00:00"},{"alias_kind":"pith_short_16","alias_value":"MXKNR2CR3XSBCG6O","created_at":"2026-07-05T00:39:05.392998+00:00"},{"alias_kind":"pith_short_8","alias_value":"MXKNR2CR","created_at":"2026-07-05T00:39:05.392998+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.16771","citing_title":"Enabling Global, Human-Centered Explanations for LLMs:From Tokens to Interpretable Code and Test Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2012.06678","citing_title":"TabTransformer: Tabular Data Modeling Using Contextual Embeddings","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C","json":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C.json","graph_json":"https://pith.science/api/pith-number/MXKNR2CR3XSBCG6OFT7UT5NJ6C/graph.json","events_json":"https://pith.science/api/pith-number/MXKNR2CR3XSBCG6OFT7UT5NJ6C/events.json","paper":"https://pith.science/paper/MXKNR2CR"},"agent_actions":{"view_html":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C","download_json":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C.json","view_paper":"https://pith.science/paper/MXKNR2CR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.04211&json=true","fetch_graph":"https://pith.science/api/pith-number/MXKNR2CR3XSBCG6OFT7UT5NJ6C/graph.json","fetch_events":"https://pith.science/api/pith-number/MXKNR2CR3XSBCG6OFT7UT5NJ6C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C/action/storage_attestation","attest_author":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C/action/author_attestation","sign_citation":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C/action/citation_signature","submit_replication":"https://pith.science/pith/MXKNR2CR3XSBCG6OFT7UT5NJ6C/action/replication_record"}},"created_at":"2026-07-05T00:39:05.392998+00:00","updated_at":"2026-07-05T00:39:05.392998+00:00"}