{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4UJQ5XURNK5O2OGFBT26YADCAQ","short_pith_number":"pith:4UJQ5XUR","schema_version":"1.0","canonical_sha256":"e5130ede916abaed38c50cf5ec00620401303212e43a9274a299d80282456c98","source":{"kind":"arxiv","id":"2309.06054","version":3},"attestation_state":"computed","paper":{"title":"Breaking through the learning plateaus of in-context learning in Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Jingwen Fu, Nanning Zheng, Tao Yang, Yan Lu, Yuwang Wang","submitted_at":"2023-09-12T08:45:25Z","abstract_excerpt":"In-context learning, i.e., learning from context examples, is an impressive ability of Transformer. Training Transformers to possess this in-context learning skill is computationally intensive due to the occurrence of learning plateaus, which are periods within the training process where there is minimal or no enhancement in the model's in-context learning capability. To study the mechanism behind the learning plateaus, we conceptually seperate a component within the model's internal representation that is exclusively affected by the model's weights. We call this the \"weights component\", and t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.06054","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-12T08:45:25Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"2b824111ec09175465d4fbf755a999feba8766b64bfe3b51060a5430b12c521d","abstract_canon_sha256":"6cdbc0ea4c745d55b969d84f8becf0b83366d105e158faf7916563fcef853ccf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:02.140175Z","signature_b64":"cXtmAKEkMIKZZ65/kZ5JXxFDX1IlygRQqusFeRG4+fLGTpXJutEyLOKxAstOLZKZBefNdu0MnApWTrFyZ8DPBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5130ede916abaed38c50cf5ec00620401303212e43a9274a299d80282456c98","last_reissued_at":"2026-07-05T08:28:02.139666Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:02.139666Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Breaking through the learning plateaus of in-context learning in Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Jingwen Fu, Nanning Zheng, Tao Yang, Yan Lu, Yuwang Wang","submitted_at":"2023-09-12T08:45:25Z","abstract_excerpt":"In-context learning, i.e., learning from context examples, is an impressive ability of Transformer. Training Transformers to possess this in-context learning skill is computationally intensive due to the occurrence of learning plateaus, which are periods within the training process where there is minimal or no enhancement in the model's in-context learning capability. To study the mechanism behind the learning plateaus, we conceptually seperate a component within the model's internal representation that is exclusively affected by the model's weights. We call this the \"weights component\", and t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.06054","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.06054/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.06054","created_at":"2026-07-05T08:28:02.139731+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.06054v3","created_at":"2026-07-05T08:28:02.139731+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.06054","created_at":"2026-07-05T08:28:02.139731+00:00"},{"alias_kind":"pith_short_12","alias_value":"4UJQ5XURNK5O","created_at":"2026-07-05T08:28:02.139731+00:00"},{"alias_kind":"pith_short_16","alias_value":"4UJQ5XURNK5O2OGF","created_at":"2026-07-05T08:28:02.139731+00:00"},{"alias_kind":"pith_short_8","alias_value":"4UJQ5XUR","created_at":"2026-07-05T08:28:02.139731+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18223","citing_title":"Instruction-as-State: Environment-Guided and State-Conditioned Semantic Understanding for Embodied Navigation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17407","citing_title":"Think before Go: Hierarchical Reasoning for Image-goal Navigation","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19064","citing_title":"The Essence of Balance for Self-Improving Agents in Vision-and-Language Navigation","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ","json":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ.json","graph_json":"https://pith.science/api/pith-number/4UJQ5XURNK5O2OGFBT26YADCAQ/graph.json","events_json":"https://pith.science/api/pith-number/4UJQ5XURNK5O2OGFBT26YADCAQ/events.json","paper":"https://pith.science/paper/4UJQ5XUR"},"agent_actions":{"view_html":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ","download_json":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ.json","view_paper":"https://pith.science/paper/4UJQ5XUR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.06054&json=true","fetch_graph":"https://pith.science/api/pith-number/4UJQ5XURNK5O2OGFBT26YADCAQ/graph.json","fetch_events":"https://pith.science/api/pith-number/4UJQ5XURNK5O2OGFBT26YADCAQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ/action/storage_attestation","attest_author":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ/action/author_attestation","sign_citation":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ/action/citation_signature","submit_replication":"https://pith.science/pith/4UJQ5XURNK5O2OGFBT26YADCAQ/action/replication_record"}},"created_at":"2026-07-05T08:28:02.139731+00:00","updated_at":"2026-07-05T08:28:02.139731+00:00"}