{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CZAYVPPZINTJR2J5RC6ZLY6FHL","short_pith_number":"pith:CZAYVPPZ","schema_version":"1.0","canonical_sha256":"16418abdf9436698e93d88bd95e3c53ac012a7516ba71bdaee12628751a54e32","source":{"kind":"arxiv","id":"2310.10616","version":1},"attestation_state":"computed","paper":{"title":"How Do Transformers Learn In-Context Beyond Simple Functions? A Case Study on Learning with Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Caiming Xiong, Huan Wang, Silvio Savarese, Song Mei, Tianyu Guo, Wei Hu, Yu Bai","submitted_at":"2023-10-16T17:40:49Z","abstract_excerpt":"While large language models based on the transformer architecture have demonstrated remarkable in-context learning (ICL) capabilities, understandings of such capabilities are still in an early stage, where existing theory and mechanistic understanding focus mostly on simple scenarios such as learning simple function classes. This paper takes initial steps on understanding ICL in more complex scenarios, by studying learning with representations. Concretely, we construct synthetic in-context learning problems with a compositional structure, where the label depends on the input through a possibly"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10616","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-16T17:40:49Z","cross_cats_sorted":[],"title_canon_sha256":"ee5e97693381344eaf59fcb8b945d8b48fbb1e5e9290cf4a18fc7547d9826c08","abstract_canon_sha256":"542f958536b2751cab9af31af02ca6e9d93ea1bf1c7ee1954d833486f32f3b73"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:22.400247Z","signature_b64":"PFv+oE47WvlDFNK8Gx0OmMrHrgkSncOSel6SRwOCrO7HIntjm/4yyDTebveUSIPgm9Kej5ieYxt/GpvEV6EHAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16418abdf9436698e93d88bd95e3c53ac012a7516ba71bdaee12628751a54e32","last_reissued_at":"2026-07-05T07:01:22.399851Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:22.399851Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Do Transformers Learn In-Context Beyond Simple Functions? A Case Study on Learning with Representations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Caiming Xiong, Huan Wang, Silvio Savarese, Song Mei, Tianyu Guo, Wei Hu, Yu Bai","submitted_at":"2023-10-16T17:40:49Z","abstract_excerpt":"While large language models based on the transformer architecture have demonstrated remarkable in-context learning (ICL) capabilities, understandings of such capabilities are still in an early stage, where existing theory and mechanistic understanding focus mostly on simple scenarios such as learning simple function classes. This paper takes initial steps on understanding ICL in more complex scenarios, by studying learning with representations. Concretely, we construct synthetic in-context learning problems with a compositional structure, where the label depends on the input through a possibly"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10616","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10616","created_at":"2026-07-05T07:01:22.399909+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10616v1","created_at":"2026-07-05T07:01:22.399909+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10616","created_at":"2026-07-05T07:01:22.399909+00:00"},{"alias_kind":"pith_short_12","alias_value":"CZAYVPPZINTJ","created_at":"2026-07-05T07:01:22.399909+00:00"},{"alias_kind":"pith_short_16","alias_value":"CZAYVPPZINTJR2J5","created_at":"2026-07-05T07:01:22.399909+00:00"},{"alias_kind":"pith_short_8","alias_value":"CZAYVPPZ","created_at":"2026-07-05T07:01:22.399909+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00605","citing_title":"Looped Transformers with Layer Normalization Provably Learn the Power Method","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05564","citing_title":"TabICL: A Tabular Foundation Model for In-Context Learning on Large Data","ref_index":203,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06609","citing_title":"Transformers Efficiently Perform In-Context Logistic Regression via Normalized Gradient Descent","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL","json":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL.json","graph_json":"https://pith.science/api/pith-number/CZAYVPPZINTJR2J5RC6ZLY6FHL/graph.json","events_json":"https://pith.science/api/pith-number/CZAYVPPZINTJR2J5RC6ZLY6FHL/events.json","paper":"https://pith.science/paper/CZAYVPPZ"},"agent_actions":{"view_html":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL","download_json":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL.json","view_paper":"https://pith.science/paper/CZAYVPPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10616&json=true","fetch_graph":"https://pith.science/api/pith-number/CZAYVPPZINTJR2J5RC6ZLY6FHL/graph.json","fetch_events":"https://pith.science/api/pith-number/CZAYVPPZINTJR2J5RC6ZLY6FHL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL/action/storage_attestation","attest_author":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL/action/author_attestation","sign_citation":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL/action/citation_signature","submit_replication":"https://pith.science/pith/CZAYVPPZINTJR2J5RC6ZLY6FHL/action/replication_record"}},"created_at":"2026-07-05T07:01:22.399909+00:00","updated_at":"2026-07-05T07:01:22.399909+00:00"}