{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KSHR2X5BJTXD7QON55ASWTR5HV","short_pith_number":"pith:KSHR2X5B","schema_version":"1.0","canonical_sha256":"548f1d5fa14cee3fc1cdef412b4e3d3d7f48b542f0a0b05c06d5e4c30fa15e8c","source":{"kind":"arxiv","id":"2502.19009","version":1},"attestation_state":"computed","paper":{"title":"Distilling Reinforcement Learning Algorithms for In-Context Model-Based Planning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gunhee Kim, Jaehyeon Son, Soochan Lee","submitted_at":"2025-02-26T10:16:57Z","abstract_excerpt":"Recent studies have shown that Transformers can perform in-context reinforcement learning (RL) by imitating existing RL algorithms, enabling sample-efficient adaptation to unseen tasks without parameter updates. However, these models also inherit the suboptimal behaviors of the RL algorithms they imitate. This issue primarily arises due to the gradual update rule employed by those algorithms. Model-based planning offers a promising solution to this limitation by allowing the models to simulate potential outcomes before taking action, providing an additional mechanism to deviate from the subopt"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.19009","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-26T10:16:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4fe9be9028471acdecff0c169db522a472ca32bf7623b405c1e08771cff2c847","abstract_canon_sha256":"4c67fa2abb6a39c9d33912c5da5a13418f54fc63f8c2cf5bef2315cf13b4d1ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:24.838986Z","signature_b64":"bedxulWhzcK9NOnxSiCCQpQpKqZWD+FLYrpojA2TkeaM6Y3TqOX1Z1LT4AOq/LE7MQJZZiBc5Q3yL4VGIAgdBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"548f1d5fa14cee3fc1cdef412b4e3d3d7f48b542f0a0b05c06d5e4c30fa15e8c","last_reissued_at":"2026-07-05T10:20:24.838526Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:24.838526Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distilling Reinforcement Learning Algorithms for In-Context Model-Based Planning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gunhee Kim, Jaehyeon Son, Soochan Lee","submitted_at":"2025-02-26T10:16:57Z","abstract_excerpt":"Recent studies have shown that Transformers can perform in-context reinforcement learning (RL) by imitating existing RL algorithms, enabling sample-efficient adaptation to unseen tasks without parameter updates. However, these models also inherit the suboptimal behaviors of the RL algorithms they imitate. This issue primarily arises due to the gradual update rule employed by those algorithms. Model-based planning offers a promising solution to this limitation by allowing the models to simulate potential outcomes before taking action, providing an additional mechanism to deviate from the subopt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.19009","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.19009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.19009","created_at":"2026-07-05T10:20:24.838581+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.19009v1","created_at":"2026-07-05T10:20:24.838581+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.19009","created_at":"2026-07-05T10:20:24.838581+00:00"},{"alias_kind":"pith_short_12","alias_value":"KSHR2X5BJTXD","created_at":"2026-07-05T10:20:24.838581+00:00"},{"alias_kind":"pith_short_16","alias_value":"KSHR2X5BJTXD7QON","created_at":"2026-07-05T10:20:24.838581+00:00"},{"alias_kind":"pith_short_8","alias_value":"KSHR2X5B","created_at":"2026-07-05T10:20:24.838581+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18812","citing_title":"Reinforcement Learning Foundation Models Should Already Be A Thing","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV","json":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV.json","graph_json":"https://pith.science/api/pith-number/KSHR2X5BJTXD7QON55ASWTR5HV/graph.json","events_json":"https://pith.science/api/pith-number/KSHR2X5BJTXD7QON55ASWTR5HV/events.json","paper":"https://pith.science/paper/KSHR2X5B"},"agent_actions":{"view_html":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV","download_json":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV.json","view_paper":"https://pith.science/paper/KSHR2X5B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.19009&json=true","fetch_graph":"https://pith.science/api/pith-number/KSHR2X5BJTXD7QON55ASWTR5HV/graph.json","fetch_events":"https://pith.science/api/pith-number/KSHR2X5BJTXD7QON55ASWTR5HV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV/action/storage_attestation","attest_author":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV/action/author_attestation","sign_citation":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV/action/citation_signature","submit_replication":"https://pith.science/pith/KSHR2X5BJTXD7QON55ASWTR5HV/action/replication_record"}},"created_at":"2026-07-05T10:20:24.838581+00:00","updated_at":"2026-07-05T10:20:24.838581+00:00"}