{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:RL5C6P7AO3XHBKBSJJYXBO44PJ","short_pith_number":"pith:RL5C6P7A","schema_version":"1.0","canonical_sha256":"8afa2f3fe076ee70a8324a7170bb9c7a7a5aecc31c283c59a05d9bc802fccd2b","source":{"kind":"arxiv","id":"2210.05675","version":2},"attestation_state":"computed","paper":{"title":"Transformers generalize differently from information stored in context vs in weights","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew K. Lampinen, Dharshan Kumaran, Felix Hill, Ishita Dasgupta, Junkyung Kim, Stephanie C.Y. Chan","submitted_at":"2022-10-11T09:29:19Z","abstract_excerpt":"Transformer models can use two fundamentally different kinds of information: information stored in weights during training, and information provided ``in-context'' at inference time. In this work, we show that transformers exhibit different inductive biases in how they represent and generalize from the information in these two sources. In particular, we characterize whether they generalize via parsimonious rules (rule-based generalization) or via direct comparison with observed examples (exemplar-based generalization). This is of important practical consequence, as it informs whether to encode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.05675","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-10-11T09:29:19Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e97a4aaee31639dfd2633a4b4c113c329e147d96a9a8be80e14990d7bfbd64cc","abstract_canon_sha256":"9ee15a78699ea59236b7f065af7900aef07ab16e2721a242232824ddbf85688f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:15.777497Z","signature_b64":"rHcCDv6ilUoP1Rn9T5hBJ19NywYEI1g1FSQ/rp4Uoy+H/lXBP4e0dz9WFjRvbA5IipaLc27Zng04gj5eJ6LvDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8afa2f3fe076ee70a8324a7170bb9c7a7a5aecc31c283c59a05d9bc802fccd2b","last_reissued_at":"2026-07-05T05:06:15.776970Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:15.776970Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers generalize differently from information stored in context vs in weights","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew K. Lampinen, Dharshan Kumaran, Felix Hill, Ishita Dasgupta, Junkyung Kim, Stephanie C.Y. Chan","submitted_at":"2022-10-11T09:29:19Z","abstract_excerpt":"Transformer models can use two fundamentally different kinds of information: information stored in weights during training, and information provided ``in-context'' at inference time. In this work, we show that transformers exhibit different inductive biases in how they represent and generalize from the information in these two sources. In particular, we characterize whether they generalize via parsimonious rules (rule-based generalization) or via direct comparison with observed examples (exemplar-based generalization). This is of important practical consequence, as it informs whether to encode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.05675","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.05675/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.05675","created_at":"2026-07-05T05:06:15.777034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.05675v2","created_at":"2026-07-05T05:06:15.777034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.05675","created_at":"2026-07-05T05:06:15.777034+00:00"},{"alias_kind":"pith_short_12","alias_value":"RL5C6P7AO3XH","created_at":"2026-07-05T05:06:15.777034+00:00"},{"alias_kind":"pith_short_16","alias_value":"RL5C6P7AO3XHBKBS","created_at":"2026-07-05T05:06:15.777034+00:00"},{"alias_kind":"pith_short_8","alias_value":"RL5C6P7A","created_at":"2026-07-05T05:06:15.777034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18164","citing_title":"Learning Dynamics of Chain-of-Thought State Tracking in a Solvable Transformer Model","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12412","citing_title":"Stories in Space: In-Context Learning Trajectories in Conceptual Belief Space","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05115","citing_title":"Manifold Steering Reveals the Shared Geometry of Neural Network Representation and Behavior","ref_index":300,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21632","citing_title":"To See the Unseen: on the Generalization Ability of Transformers in Symbolic Reasoning","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ","json":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ.json","graph_json":"https://pith.science/api/pith-number/RL5C6P7AO3XHBKBSJJYXBO44PJ/graph.json","events_json":"https://pith.science/api/pith-number/RL5C6P7AO3XHBKBSJJYXBO44PJ/events.json","paper":"https://pith.science/paper/RL5C6P7A"},"agent_actions":{"view_html":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ","download_json":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ.json","view_paper":"https://pith.science/paper/RL5C6P7A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.05675&json=true","fetch_graph":"https://pith.science/api/pith-number/RL5C6P7AO3XHBKBSJJYXBO44PJ/graph.json","fetch_events":"https://pith.science/api/pith-number/RL5C6P7AO3XHBKBSJJYXBO44PJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ/action/storage_attestation","attest_author":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ/action/author_attestation","sign_citation":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ/action/citation_signature","submit_replication":"https://pith.science/pith/RL5C6P7AO3XHBKBSJJYXBO44PJ/action/replication_record"}},"created_at":"2026-07-05T05:06:15.777034+00:00","updated_at":"2026-07-05T05:06:15.777034+00:00"}