{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MIXKGVVUOVEEWLYKHF3BGCBYMD","short_pith_number":"pith:MIXKGVVU","schema_version":"1.0","canonical_sha256":"622ea356b475484b2f0a397613083860e0f0bfb52d502f9cc782aa2143566b53","source":{"kind":"arxiv","id":"2408.09503","version":2},"attestation_state":"computed","paper":{"title":"Out-of-distribution generalization via composition: a lens through induction heads in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Jiajun Song, Yiqiao Zhong, Zhuoyan Xu","submitted_at":"2024-08-18T14:52:25Z","abstract_excerpt":"Large language models (LLMs) such as GPT-4 sometimes appear to be creative, solving novel tasks often with a few demonstrations in the prompt. These tasks require the models to generalize on distributions different from those from training data -- which is known as out-of-distribution (OOD) generalization. Despite the tremendous success of LLMs, how they approach OOD generalization remains an open and underexplored question. We examine OOD generalization in settings where instances are generated according to hidden rules, including in-context learning with symbolic reasoning. Models are requir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.09503","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-18T14:52:25Z","cross_cats_sorted":["cs.AI","cs.LG","stat.ML"],"title_canon_sha256":"84edea9f3a11f91fcaddd99aeff3344d15e6636afbcbfcb0c664e26589fafc36","abstract_canon_sha256":"ab6e8bb35f09823ec841200ec1bf26e110e1b99e6d7dedc0da777b533cb24bab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:46.371264Z","signature_b64":"dIcT3QZv0QEGtOIUa5TjKc961W7i4/3AHLjIfQKb/2vSp4CLJcWs/UYWE/AMhjNnb2pIemPdt0j3+aS1b3+BBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"622ea356b475484b2f0a397613083860e0f0bfb52d502f9cc782aa2143566b53","last_reissued_at":"2026-07-05T09:54:46.370777Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:46.370777Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Out-of-distribution generalization via composition: a lens through induction heads in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Jiajun Song, Yiqiao Zhong, Zhuoyan Xu","submitted_at":"2024-08-18T14:52:25Z","abstract_excerpt":"Large language models (LLMs) such as GPT-4 sometimes appear to be creative, solving novel tasks often with a few demonstrations in the prompt. These tasks require the models to generalize on distributions different from those from training data -- which is known as out-of-distribution (OOD) generalization. Despite the tremendous success of LLMs, how they approach OOD generalization remains an open and underexplored question. We examine OOD generalization in settings where instances are generated according to hidden rules, including in-context learning with symbolic reasoning. Models are requir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.09503","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.09503/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.09503","created_at":"2026-07-05T09:54:46.370836+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.09503v2","created_at":"2026-07-05T09:54:46.370836+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.09503","created_at":"2026-07-05T09:54:46.370836+00:00"},{"alias_kind":"pith_short_12","alias_value":"MIXKGVVUOVEE","created_at":"2026-07-05T09:54:46.370836+00:00"},{"alias_kind":"pith_short_16","alias_value":"MIXKGVVUOVEEWLYK","created_at":"2026-07-05T09:54:46.370836+00:00"},{"alias_kind":"pith_short_8","alias_value":"MIXKGVVU","created_at":"2026-07-05T09:54:46.370836+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24164","citing_title":"Localizing Task Recognition and Task Learning in In-Context Learning via Attention Head Analysis","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD","json":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD.json","graph_json":"https://pith.science/api/pith-number/MIXKGVVUOVEEWLYKHF3BGCBYMD/graph.json","events_json":"https://pith.science/api/pith-number/MIXKGVVUOVEEWLYKHF3BGCBYMD/events.json","paper":"https://pith.science/paper/MIXKGVVU"},"agent_actions":{"view_html":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD","download_json":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD.json","view_paper":"https://pith.science/paper/MIXKGVVU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.09503&json=true","fetch_graph":"https://pith.science/api/pith-number/MIXKGVVUOVEEWLYKHF3BGCBYMD/graph.json","fetch_events":"https://pith.science/api/pith-number/MIXKGVVUOVEEWLYKHF3BGCBYMD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD/action/storage_attestation","attest_author":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD/action/author_attestation","sign_citation":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD/action/citation_signature","submit_replication":"https://pith.science/pith/MIXKGVVUOVEEWLYKHF3BGCBYMD/action/replication_record"}},"created_at":"2026-07-05T09:54:46.370836+00:00","updated_at":"2026-07-05T09:54:46.370836+00:00"}