{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SBUYY6LZBDDQFZSTCKDFIZF5WU","short_pith_number":"pith:SBUYY6LZ","schema_version":"1.0","canonical_sha256":"90698c797908c702e65312865464bdb505422c20bda301f5e26d1acbb2a53059","source":{"kind":"arxiv","id":"2302.00294","version":2},"attestation_state":"computed","paper":{"title":"The geometry of hidden representations of large transformer models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alberto Cazzaniga, Alessandro Laio, Alessio Ansuini, Diego Doimo, Francesca Cuturello, Lucrezia Valeriani","submitted_at":"2023-02-01T07:50:26Z","abstract_excerpt":"Large transformers are powerful architectures used for self-supervised data analysis across various data types, including protein sequences, images, and text. In these models, the semantic structure of the dataset emerges from a sequence of transformations between one representation and the next. We characterize the geometric and statistical properties of these representations and how they change as we move through the layers. By analyzing the intrinsic dimension (ID) and neighbor composition, we find that the representations evolve similarly in transformers trained on protein language tasks a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.00294","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-01T07:50:26Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"37e1c66e0985f6c9e3d34c0217dc60d2fa41fe11df438c3828f4bf5a0ff9a6b4","abstract_canon_sha256":"7f3371ef950dbb028fd0c71202e4793dc89f9779f14834019c6112540f616862"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:22.855728Z","signature_b64":"ywzjedqdw209nvVFGFscY4ZBkgpCXHsf7JlBDPjEyzLAxbDljNbRX42ajukxdxV111EqaGKUFiwXYv0GyTBrBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90698c797908c702e65312865464bdb505422c20bda301f5e26d1acbb2a53059","last_reissued_at":"2026-07-05T07:06:22.855248Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:22.855248Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The geometry of hidden representations of large transformer models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alberto Cazzaniga, Alessandro Laio, Alessio Ansuini, Diego Doimo, Francesca Cuturello, Lucrezia Valeriani","submitted_at":"2023-02-01T07:50:26Z","abstract_excerpt":"Large transformers are powerful architectures used for self-supervised data analysis across various data types, including protein sequences, images, and text. In these models, the semantic structure of the dataset emerges from a sequence of transformations between one representation and the next. We characterize the geometric and statistical properties of these representations and how they change as we move through the layers. By analyzing the intrinsic dimension (ID) and neighbor composition, we find that the representations evolve similarly in transformers trained on protein language tasks a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.00294","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.00294/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.00294","created_at":"2026-07-05T07:06:22.855305+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.00294v2","created_at":"2026-07-05T07:06:22.855305+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.00294","created_at":"2026-07-05T07:06:22.855305+00:00"},{"alias_kind":"pith_short_12","alias_value":"SBUYY6LZBDDQ","created_at":"2026-07-05T07:06:22.855305+00:00"},{"alias_kind":"pith_short_16","alias_value":"SBUYY6LZBDDQFZST","created_at":"2026-07-05T07:06:22.855305+00:00"},{"alias_kind":"pith_short_8","alias_value":"SBUYY6LZ","created_at":"2026-07-05T07:06:22.855305+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08321","citing_title":"Texture Representations in Deep Vision Models: Comparing CNNs, Vision Transformers, and Human Perception","ref_index":60,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07047","citing_title":"Riemannian Geometry for Pre-trained Language Model Embeddings","ref_index":54,"is_internal_anchor":true},{"citing_arxiv_id":"2604.19321","citing_title":"RDP LoRA: Geometry-Driven Identification for Parameter-Efficient Adaptation in Large Language Models","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU","json":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU.json","graph_json":"https://pith.science/api/pith-number/SBUYY6LZBDDQFZSTCKDFIZF5WU/graph.json","events_json":"https://pith.science/api/pith-number/SBUYY6LZBDDQFZSTCKDFIZF5WU/events.json","paper":"https://pith.science/paper/SBUYY6LZ"},"agent_actions":{"view_html":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU","download_json":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU.json","view_paper":"https://pith.science/paper/SBUYY6LZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.00294&json=true","fetch_graph":"https://pith.science/api/pith-number/SBUYY6LZBDDQFZSTCKDFIZF5WU/graph.json","fetch_events":"https://pith.science/api/pith-number/SBUYY6LZBDDQFZSTCKDFIZF5WU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU/action/storage_attestation","attest_author":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU/action/author_attestation","sign_citation":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU/action/citation_signature","submit_replication":"https://pith.science/pith/SBUYY6LZBDDQFZSTCKDFIZF5WU/action/replication_record"}},"created_at":"2026-07-05T07:06:22.855305+00:00","updated_at":"2026-07-05T07:06:22.855305+00:00"}