{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:MKZL6UZVDXA5IODWRZIH2NJDZR","short_pith_number":"pith:MKZL6UZV","schema_version":"1.0","canonical_sha256":"62b2bf53351dc1d438768e507d3523cc684a4751a44c648e0b8c8791ed160f28","source":{"kind":"arxiv","id":"2205.05055","version":6},"attestation_state":"computed","paper":{"title":"Data Distributional Properties Drive Emergent In-Context Learning in Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aaditya Singh, Adam Santoro, Andrew K. Lampinen, Felix Hill, Jane X. Wang, Jay McClelland, Pierre H. Richemond, Stephanie C.Y. Chan","submitted_at":"2022-04-22T16:10:50Z","abstract_excerpt":"Large transformer-based models are able to perform in-context few-shot learning, without being explicitly trained for it. This observation raises the question: what aspects of the training regime lead to this emergent behavior? Here, we show that this behavior is driven by the distributions of the training data itself. In-context learning emerges when the training data exhibits particular distributional properties such as burstiness (items appear in clusters rather than being uniformly distributed over time) and having large numbers of rarely occurring classes. In-context learning also emerges"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.05055","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-22T16:10:50Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"44a73d1cdd88c627737d11e4a77217e10c59be5e161fa8a15f035e6b47bd5dee","abstract_canon_sha256":"dbab8a5debd6840156cb6601fce026e88deed56c2835122a41a32d337e0375ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:17:01.926908Z","signature_b64":"cDi/0jzled23XZHpmvZD1E3axvnWzdIQ1YUTNSJ3+G8HrR8164Vk7Hr+T5JkxE7lENHC4GOsUsargVxY964iAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62b2bf53351dc1d438768e507d3523cc684a4751a44c648e0b8c8791ed160f28","last_reissued_at":"2026-07-05T05:17:01.926451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:17:01.926451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Data Distributional Properties Drive Emergent In-Context Learning in Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aaditya Singh, Adam Santoro, Andrew K. Lampinen, Felix Hill, Jane X. Wang, Jay McClelland, Pierre H. Richemond, Stephanie C.Y. Chan","submitted_at":"2022-04-22T16:10:50Z","abstract_excerpt":"Large transformer-based models are able to perform in-context few-shot learning, without being explicitly trained for it. This observation raises the question: what aspects of the training regime lead to this emergent behavior? Here, we show that this behavior is driven by the distributions of the training data itself. In-context learning emerges when the training data exhibits particular distributional properties such as burstiness (items appear in clusters rather than being uniformly distributed over time) and having large numbers of rarely occurring classes. In-context learning also emerges"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.05055","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.05055/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.05055","created_at":"2026-07-05T05:17:01.926508+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.05055v6","created_at":"2026-07-05T05:17:01.926508+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.05055","created_at":"2026-07-05T05:17:01.926508+00:00"},{"alias_kind":"pith_short_12","alias_value":"MKZL6UZVDXA5","created_at":"2026-07-05T05:17:01.926508+00:00"},{"alias_kind":"pith_short_16","alias_value":"MKZL6UZVDXA5IODW","created_at":"2026-07-05T05:17:01.926508+00:00"},{"alias_kind":"pith_short_8","alias_value":"MKZL6UZV","created_at":"2026-07-05T05:17:01.926508+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01033","citing_title":"The Model Organism Lottery: Model Organism Interpretability Strongly Depends on Training Methodology","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14130","citing_title":"PRISM: Prompt-Refined In-Context System Modelling for Financial Retrieval","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2211.15661","citing_title":"What learning algorithm is in-context learning? Investigations with linear models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2206.07682","citing_title":"Emergent Abilities of Large Language Models","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR","json":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR.json","graph_json":"https://pith.science/api/pith-number/MKZL6UZVDXA5IODWRZIH2NJDZR/graph.json","events_json":"https://pith.science/api/pith-number/MKZL6UZVDXA5IODWRZIH2NJDZR/events.json","paper":"https://pith.science/paper/MKZL6UZV"},"agent_actions":{"view_html":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR","download_json":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR.json","view_paper":"https://pith.science/paper/MKZL6UZV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.05055&json=true","fetch_graph":"https://pith.science/api/pith-number/MKZL6UZVDXA5IODWRZIH2NJDZR/graph.json","fetch_events":"https://pith.science/api/pith-number/MKZL6UZVDXA5IODWRZIH2NJDZR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR/action/storage_attestation","attest_author":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR/action/author_attestation","sign_citation":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR/action/citation_signature","submit_replication":"https://pith.science/pith/MKZL6UZVDXA5IODWRZIH2NJDZR/action/replication_record"}},"created_at":"2026-07-05T05:17:01.926508+00:00","updated_at":"2026-07-05T05:17:01.926508+00:00"}