{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RSXCFYBPGQW4AHU7WSTL2KXTUK","short_pith_number":"pith:RSXCFYBP","schema_version":"1.0","canonical_sha256":"8cae22e02f342dc01e9fb4a6bd2af3a29302b9c63b4e52777b6541ab923d1370","source":{"kind":"arxiv","id":"2310.19956","version":2},"attestation_state":"computed","paper":{"title":"The Impact of Depth on Compositional Generalization in Transformer Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dan Garrette, Fei Sha, Ishita Dasgupta, Jackson Petty, Sjoerd van Steenkiste, Tal Linzen","submitted_at":"2023-10-30T19:10:06Z","abstract_excerpt":"To process novel sentences, language models (LMs) must generalize compositionally -- combine familiar elements in new ways. What aspects of a model's structure promote compositional generalization? Focusing on transformers, we test the hypothesis, motivated by theoretical and empirical work, that deeper transformers generalize more compositionally. Simply adding layers increases the total number of parameters; to address this confound between depth and size, we construct three classes of models which trade off depth for width such that the total number of parameters is kept constant (41M, 134M"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.19956","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-30T19:10:06Z","cross_cats_sorted":[],"title_canon_sha256":"70f3ef084f4b08dcd892ebfa50eeecb91d79093b608a866f18162ca2769e33c9","abstract_canon_sha256":"dab2ee3b749111469c4232013fb11781181d3a3b0e8b56244b3e8e59c4af25a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:43.960431Z","signature_b64":"FB2bXJxMe+2fnfH1u0OH1zBR/XpHqDpG8cvsWc3NabNzs8bj+sXO2Z2WB++51QgIkA/PZ8WbmvgTmc6jntCLAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8cae22e02f342dc01e9fb4a6bd2af3a29302b9c63b4e52777b6541ab923d1370","last_reissued_at":"2026-07-05T08:06:43.959934Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:43.959934Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Impact of Depth on Compositional Generalization in Transformer Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dan Garrette, Fei Sha, Ishita Dasgupta, Jackson Petty, Sjoerd van Steenkiste, Tal Linzen","submitted_at":"2023-10-30T19:10:06Z","abstract_excerpt":"To process novel sentences, language models (LMs) must generalize compositionally -- combine familiar elements in new ways. What aspects of a model's structure promote compositional generalization? Focusing on transformers, we test the hypothesis, motivated by theoretical and empirical work, that deeper transformers generalize more compositionally. Simply adding layers increases the total number of parameters; to address this confound between depth and size, we construct three classes of models which trade off depth for width such that the total number of parameters is kept constant (41M, 134M"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19956","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19956/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.19956","created_at":"2026-07-05T08:06:43.959992+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.19956v2","created_at":"2026-07-05T08:06:43.959992+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19956","created_at":"2026-07-05T08:06:43.959992+00:00"},{"alias_kind":"pith_short_12","alias_value":"RSXCFYBPGQW4","created_at":"2026-07-05T08:06:43.959992+00:00"},{"alias_kind":"pith_short_16","alias_value":"RSXCFYBPGQW4AHU7","created_at":"2026-07-05T08:06:43.959992+00:00"},{"alias_kind":"pith_short_8","alias_value":"RSXCFYBP","created_at":"2026-07-05T08:06:43.959992+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10933","citing_title":"DECO: Sparse Mixture-of-Experts with Dense-Comparable Performance on End-Side Devices","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18245","citing_title":"Scaling Laws Meet Model Architecture: Toward Inference-Efficient LLMs","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10933","citing_title":"DECO: Sparse Mixture-of-Experts with Dense-Comparable Performance on End-Side Devices","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10933","citing_title":"DECO: Sparse Mixture-of-Experts with Dense-Comparable Performance on End-Side Devices","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15306","citing_title":"Generalization in LLM Problem Solving: The Case of the Shortest Path","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK","json":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK.json","graph_json":"https://pith.science/api/pith-number/RSXCFYBPGQW4AHU7WSTL2KXTUK/graph.json","events_json":"https://pith.science/api/pith-number/RSXCFYBPGQW4AHU7WSTL2KXTUK/events.json","paper":"https://pith.science/paper/RSXCFYBP"},"agent_actions":{"view_html":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK","download_json":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK.json","view_paper":"https://pith.science/paper/RSXCFYBP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.19956&json=true","fetch_graph":"https://pith.science/api/pith-number/RSXCFYBPGQW4AHU7WSTL2KXTUK/graph.json","fetch_events":"https://pith.science/api/pith-number/RSXCFYBPGQW4AHU7WSTL2KXTUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK/action/storage_attestation","attest_author":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK/action/author_attestation","sign_citation":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK/action/citation_signature","submit_replication":"https://pith.science/pith/RSXCFYBPGQW4AHU7WSTL2KXTUK/action/replication_record"}},"created_at":"2026-07-05T08:06:43.959992+00:00","updated_at":"2026-07-05T08:06:43.959992+00:00"}