{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU","short_pith_number":"pith:FCA3Z4ZO","schema_version":"1.0","canonical_sha256":"2881bcf32e4ee11cf4d7d9dc15f7150d022ea7760a9daa49a8f9b45c8054a1b3","source":{"kind":"arxiv","id":"2405.13861","version":4},"attestation_state":"computed","paper":{"title":"Transformers Can Learn Temporal Difference Methods for In-Context Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ethan Blaser, Hadi Daneshmand, Jiuqi Wang, Shangtong Zhang","submitted_at":"2024-05-22T17:38:16Z","abstract_excerpt":"Traditionally, reinforcement learning (RL) agents learn to solve new tasks by updating their neural network parameters through interactions with the task environment. However, recent works demonstrate that some RL agents, after certain pretraining procedures, can learn to solve unseen new tasks without parameter updates, a phenomenon known as in-context reinforcement learning (ICRL). The empirical success of ICRL is widely attributed to the hypothesis that the forward pass of the pretrained agent neural network implements an RL algorithm. In this paper, we support this hypothesis by showing, b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13861","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-22T17:38:16Z","cross_cats_sorted":[],"title_canon_sha256":"bf291451d38a8ede4b9c4eb61197a9a74eb57eb10317df64639dfbe2376233c2","abstract_canon_sha256":"16aefcefbc7c6f54b14c80401e8e69732cb52951d782a2c001e0de46b3b7084e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:24.178522Z","signature_b64":"AXGRvA38TwEbe1qE1gLMW//5evRCliZsg08MqhpBJ9rwIrG38674Kp9mgW9JgPVooYIJQSnDup+TDVb3t1VUBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2881bcf32e4ee11cf4d7d9dc15f7150d022ea7760a9daa49a8f9b45c8054a1b3","last_reissued_at":"2026-07-05T10:19:24.177887Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:24.177887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers Can Learn Temporal Difference Methods for In-Context Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ethan Blaser, Hadi Daneshmand, Jiuqi Wang, Shangtong Zhang","submitted_at":"2024-05-22T17:38:16Z","abstract_excerpt":"Traditionally, reinforcement learning (RL) agents learn to solve new tasks by updating their neural network parameters through interactions with the task environment. However, recent works demonstrate that some RL agents, after certain pretraining procedures, can learn to solve unseen new tasks without parameter updates, a phenomenon known as in-context reinforcement learning (ICRL). The empirical success of ICRL is widely attributed to the hypothesis that the forward pass of the pretrained agent neural network implements an RL algorithm. In this paper, we support this hypothesis by showing, b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13861","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13861/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13861","created_at":"2026-07-05T10:19:24.177948+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13861v4","created_at":"2026-07-05T10:19:24.177948+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13861","created_at":"2026-07-05T10:19:24.177948+00:00"},{"alias_kind":"pith_short_12","alias_value":"FCA3Z4ZOJ3QR","created_at":"2026-07-05T10:19:24.177948+00:00"},{"alias_kind":"pith_short_16","alias_value":"FCA3Z4ZOJ3QRZ5GX","created_at":"2026-07-05T10:19:24.177948+00:00"},{"alias_kind":"pith_short_8","alias_value":"FCA3Z4ZO","created_at":"2026-07-05T10:19:24.177948+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29714","citing_title":"UniVAD v2: Unified Visual Anomaly Detection via Support-Conditioned Boundary Construction","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09727","citing_title":"One for All: A Non-Linear Transformer can Enable Cross-Domain Generalization for In-Context Reinforcement Learning","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU","json":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU.json","graph_json":"https://pith.science/api/pith-number/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/graph.json","events_json":"https://pith.science/api/pith-number/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/events.json","paper":"https://pith.science/paper/FCA3Z4ZO"},"agent_actions":{"view_html":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU","download_json":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU.json","view_paper":"https://pith.science/paper/FCA3Z4ZO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13861&json=true","fetch_graph":"https://pith.science/api/pith-number/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/graph.json","fetch_events":"https://pith.science/api/pith-number/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/action/storage_attestation","attest_author":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/action/author_attestation","sign_citation":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/action/citation_signature","submit_replication":"https://pith.science/pith/FCA3Z4ZOJ3QRZ5GX3HOBL5YVBU/action/replication_record"}},"created_at":"2026-07-05T10:19:24.177948+00:00","updated_at":"2026-07-05T10:19:24.177948+00:00"}