{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CVIQCQIYEB32HWCJXMQ6S54DXU","short_pith_number":"pith:CVIQCQIY","schema_version":"1.0","canonical_sha256":"15510141182077a3d849bb21e97783bd1ebb4b45b6318dd141c52b53c9484ef3","source":{"kind":"arxiv","id":"2502.07978","version":1},"attestation_state":"computed","paper":{"title":"A Survey of In-Context Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amir Moeini, Ethan Blaser, Jacob Beck, Jiuqi Wang, Rohan Chandra, Shangtong Zhang, Shimon Whiteson","submitted_at":"2025-02-11T21:52:19Z","abstract_excerpt":"Reinforcement learning (RL) agents typically optimize their policies by performing expensive backward passes to update their network parameters. However, some agents can solve new tasks without updating any parameters by simply conditioning on additional context such as their action-observation histories. This paper surveys work on such behavior, known as in-context reinforcement learning."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.07978","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-11T21:52:19Z","cross_cats_sorted":[],"title_canon_sha256":"73066347c6132f9a45b82ce6050c8c91b42ea9ad283f466a3c7a7be099e70ffd","abstract_canon_sha256":"9065d3049b3c70dad84f3209613e490534de89dda2f46bdf9492d97cea0e4c57"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:13.093300Z","signature_b64":"qEZdnU01/9c+o8iBmhzq8VRWnxFkDQJgghTGbSIw5adBIrOm6cpxUJW3nwSEnaxLGxPT7Y/JVGJkS+/O+wRwDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15510141182077a3d849bb21e97783bd1ebb4b45b6318dd141c52b53c9484ef3","last_reissued_at":"2026-07-05T10:13:13.092805Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:13.092805Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of In-Context Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amir Moeini, Ethan Blaser, Jacob Beck, Jiuqi Wang, Rohan Chandra, Shangtong Zhang, Shimon Whiteson","submitted_at":"2025-02-11T21:52:19Z","abstract_excerpt":"Reinforcement learning (RL) agents typically optimize their policies by performing expensive backward passes to update their network parameters. However, some agents can solve new tasks without updating any parameters by simply conditioning on additional context such as their action-observation histories. This paper surveys work on such behavior, known as in-context reinforcement learning."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.07978","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.07978/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.07978","created_at":"2026-07-05T10:13:13.092862+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.07978v1","created_at":"2026-07-05T10:13:13.092862+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.07978","created_at":"2026-07-05T10:13:13.092862+00:00"},{"alias_kind":"pith_short_12","alias_value":"CVIQCQIYEB32","created_at":"2026-07-05T10:13:13.092862+00:00"},{"alias_kind":"pith_short_16","alias_value":"CVIQCQIYEB32HWCJ","created_at":"2026-07-05T10:13:13.092862+00:00"},{"alias_kind":"pith_short_8","alias_value":"CVIQCQIY","created_at":"2026-07-05T10:13:13.092862+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24423","citing_title":"Benchmarking the Limits of In-Context Reinforcement Learning for Ad-Hoc Teamwork","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17431","citing_title":"MATE: Solving Contextual Markov Decision Processes with Memory of Accumulated Transition Embeddings","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14274","citing_title":"Discovering New Theorems via LLMs with In-Context Proof Learning in Lean","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09727","citing_title":"One for All: A Non-Linear Transformer can Enable Cross-Domain Generalization for In-Context Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05859","citing_title":"When Do We Need LLMs? A Diagnostic for Language-Driven Bandits","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05429","citing_title":"Bridging Natural Language and Microgrid Dynamics: A Context-Aware Simulator and Dataset","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU","json":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU.json","graph_json":"https://pith.science/api/pith-number/CVIQCQIYEB32HWCJXMQ6S54DXU/graph.json","events_json":"https://pith.science/api/pith-number/CVIQCQIYEB32HWCJXMQ6S54DXU/events.json","paper":"https://pith.science/paper/CVIQCQIY"},"agent_actions":{"view_html":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU","download_json":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU.json","view_paper":"https://pith.science/paper/CVIQCQIY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.07978&json=true","fetch_graph":"https://pith.science/api/pith-number/CVIQCQIYEB32HWCJXMQ6S54DXU/graph.json","fetch_events":"https://pith.science/api/pith-number/CVIQCQIYEB32HWCJXMQ6S54DXU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU/action/storage_attestation","attest_author":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU/action/author_attestation","sign_citation":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU/action/citation_signature","submit_replication":"https://pith.science/pith/CVIQCQIYEB32HWCJXMQ6S54DXU/action/replication_record"}},"created_at":"2026-07-05T10:13:13.092862+00:00","updated_at":"2026-07-05T10:13:13.092862+00:00"}