{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DK4HEZEEIUG7IOJU24BRZIVAOO","short_pith_number":"pith:DK4HEZEE","schema_version":"1.0","canonical_sha256":"1ab8726484450df43934d7031ca2a07382021543dc4f6514590aaf9d35894d94","source":{"kind":"arxiv","id":"2407.16153","version":1},"attestation_state":"computed","paper":{"title":"On the Benefits of Rank in Attention Layers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Gilad Yehudai, Joan Bruna, Noah Amsel","submitted_at":"2024-07-23T03:40:24Z","abstract_excerpt":"Attention-based mechanisms are widely used in machine learning, most prominently in transformers. However, hyperparameters such as the rank of the attention matrices and the number of heads are scaled nearly the same way in all realizations of this architecture, without theoretical justification. In this work we show that there are dramatic trade-offs between the rank and number of heads of the attention mechanism. Specifically, we present a simple and natural target function that can be represented using a single full-rank attention head for any context length, but that cannot be approximated"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.16153","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-23T03:40:24Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"fdf8e9f5fab4ceccf52f1bc309415b0f8c3ac38046cad165cfce25b3935f3f0f","abstract_canon_sha256":"612fff539d8de5119106717e3c00380bef78871bb4b4f022238063ea86e8d997"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:47:27.949041Z","signature_b64":"50/9VyZdJwW1xDZ9HngEJHMfoJQEKV7CP/E/uHK2lLxlhNYix/P6cEPLeZQXIwoD5oh6LLDNqxFqpfwTQfU/AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1ab8726484450df43934d7031ca2a07382021543dc4f6514590aaf9d35894d94","last_reissued_at":"2026-07-05T08:47:27.948634Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:47:27.948634Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Benefits of Rank in Attention Layers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Gilad Yehudai, Joan Bruna, Noah Amsel","submitted_at":"2024-07-23T03:40:24Z","abstract_excerpt":"Attention-based mechanisms are widely used in machine learning, most prominently in transformers. However, hyperparameters such as the rank of the attention matrices and the number of heads are scaled nearly the same way in all realizations of this architecture, without theoretical justification. In this work we show that there are dramatic trade-offs between the rank and number of heads of the attention mechanism. Specifically, we present a simple and natural target function that can be represented using a single full-rank attention head for any context length, but that cannot be approximated"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.16153","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.16153/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.16153","created_at":"2026-07-05T08:47:27.948694+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.16153v1","created_at":"2026-07-05T08:47:27.948694+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.16153","created_at":"2026-07-05T08:47:27.948694+00:00"},{"alias_kind":"pith_short_12","alias_value":"DK4HEZEEIUG7","created_at":"2026-07-05T08:47:27.948694+00:00"},{"alias_kind":"pith_short_16","alias_value":"DK4HEZEEIUG7IOJU","created_at":"2026-07-05T08:47:27.948694+00:00"},{"alias_kind":"pith_short_8","alias_value":"DK4HEZEE","created_at":"2026-07-05T08:47:27.948694+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.07963","citing_title":"Customizing the Inductive Biases of Softmax Attention using Structured Matrices","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO","json":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO.json","graph_json":"https://pith.science/api/pith-number/DK4HEZEEIUG7IOJU24BRZIVAOO/graph.json","events_json":"https://pith.science/api/pith-number/DK4HEZEEIUG7IOJU24BRZIVAOO/events.json","paper":"https://pith.science/paper/DK4HEZEE"},"agent_actions":{"view_html":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO","download_json":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO.json","view_paper":"https://pith.science/paper/DK4HEZEE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.16153&json=true","fetch_graph":"https://pith.science/api/pith-number/DK4HEZEEIUG7IOJU24BRZIVAOO/graph.json","fetch_events":"https://pith.science/api/pith-number/DK4HEZEEIUG7IOJU24BRZIVAOO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO/action/storage_attestation","attest_author":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO/action/author_attestation","sign_citation":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO/action/citation_signature","submit_replication":"https://pith.science/pith/DK4HEZEEIUG7IOJU24BRZIVAOO/action/replication_record"}},"created_at":"2026-07-05T08:47:27.948694+00:00","updated_at":"2026-07-05T08:47:27.948694+00:00"}