{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BT226H54T56PJ35VWUWJTLWMKX","short_pith_number":"pith:BT226H54","schema_version":"1.0","canonical_sha256":"0cf5af1fbc9f7cf4efb5b52c99aecc55ce30f288d302e185743aa6d8d769910b","source":{"kind":"arxiv","id":"2508.19344","version":1},"attestation_state":"computed","paper":{"title":"Re:Frame -- Retrieving Experience From Associative Memory","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aleksandr I. Panov, Alexey K. Kovalev, Daniil Zelezetsky, Egor Cherepanov","submitted_at":"2025-08-26T18:05:09Z","abstract_excerpt":"Offline reinforcement learning (RL) often deals with suboptimal data when collecting large expert datasets is unavailable or impractical. This limitation makes it difficult for agents to generalize and achieve high performance, as they must learn primarily from imperfect or inconsistent trajectories. A central challenge is therefore how to best leverage scarce expert demonstrations alongside abundant but lower-quality data. We demonstrate that incorporating even a tiny amount of expert experience can substantially improve RL agent performance. We introduce Re:Frame (Retrieving Experience From "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.19344","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-26T18:05:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a5e1557ed5b68f84d42c093535a14eb9db96e90caf564d06191a333e808ab5fd","abstract_canon_sha256":"197fe5c3e5192d6152387fbbc15f11743918a4ffe8c45c0a9965011eaff4002d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:43.592556Z","signature_b64":"y1EEJ4ZS3kDTqJ5RD67pyVEt4n3QKX/18LPWUHHBYIKgntjF7Q5ieGuXcWHuiM8Z/g4cMm7O/a50pAmOkW4EBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0cf5af1fbc9f7cf4efb5b52c99aecc55ce30f288d302e185743aa6d8d769910b","last_reissued_at":"2026-07-05T11:59:43.592080Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:43.592080Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Re:Frame -- Retrieving Experience From Associative Memory","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aleksandr I. Panov, Alexey K. Kovalev, Daniil Zelezetsky, Egor Cherepanov","submitted_at":"2025-08-26T18:05:09Z","abstract_excerpt":"Offline reinforcement learning (RL) often deals with suboptimal data when collecting large expert datasets is unavailable or impractical. This limitation makes it difficult for agents to generalize and achieve high performance, as they must learn primarily from imperfect or inconsistent trajectories. A central challenge is therefore how to best leverage scarce expert demonstrations alongside abundant but lower-quality data. We demonstrate that incorporating even a tiny amount of expert experience can substantially improve RL agent performance. We introduce Re:Frame (Retrieving Experience From "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.19344","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.19344/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.19344","created_at":"2026-07-05T11:59:43.592139+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.19344v1","created_at":"2026-07-05T11:59:43.592139+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.19344","created_at":"2026-07-05T11:59:43.592139+00:00"},{"alias_kind":"pith_short_12","alias_value":"BT226H54T56P","created_at":"2026-07-05T11:59:43.592139+00:00"},{"alias_kind":"pith_short_16","alias_value":"BT226H54T56PJ35V","created_at":"2026-07-05T11:59:43.592139+00:00"},{"alias_kind":"pith_short_8","alias_value":"BT226H54","created_at":"2026-07-05T11:59:43.592139+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX","json":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX.json","graph_json":"https://pith.science/api/pith-number/BT226H54T56PJ35VWUWJTLWMKX/graph.json","events_json":"https://pith.science/api/pith-number/BT226H54T56PJ35VWUWJTLWMKX/events.json","paper":"https://pith.science/paper/BT226H54"},"agent_actions":{"view_html":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX","download_json":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX.json","view_paper":"https://pith.science/paper/BT226H54","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.19344&json=true","fetch_graph":"https://pith.science/api/pith-number/BT226H54T56PJ35VWUWJTLWMKX/graph.json","fetch_events":"https://pith.science/api/pith-number/BT226H54T56PJ35VWUWJTLWMKX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX/action/storage_attestation","attest_author":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX/action/author_attestation","sign_citation":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX/action/citation_signature","submit_replication":"https://pith.science/pith/BT226H54T56PJ35VWUWJTLWMKX/action/replication_record"}},"created_at":"2026-07-05T11:59:43.592139+00:00","updated_at":"2026-07-05T11:59:43.592139+00:00"}