{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:XFRLHEB5KVWRYUIPE47OJV4RZU","short_pith_number":"pith:XFRLHEB5","canonical_record":{"source":{"id":"2412.07762","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-10T18:57:12Z","cross_cats_sorted":[],"title_canon_sha256":"01f6188ae65f25c9bb7ef8309d5cd352d57c58e56e551cebca37dc39f33ef182","abstract_canon_sha256":"df614d70855dabc961cc5e71d300ea04b6a190951bd245ceb06c56126c238944"},"schema_version":"1.0"},"canonical_sha256":"b962b3903d556d1c510f273ee4d791cd2c3ccea27190b0cdda0e8b2398e67cdb","source":{"kind":"arxiv","id":"2412.07762","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2412.07762","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"arxiv_version","alias_value":"2412.07762v3","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.07762","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"pith_short_12","alias_value":"XFRLHEB5KVWR","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"pith_short_16","alias_value":"XFRLHEB5KVWRYUIP","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"pith_short_8","alias_value":"XFRLHEB5","created_at":"2026-07-05T11:30:39Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:XFRLHEB5KVWRYUIPE47OJV4RZU","target":"record","payload":{"canonical_record":{"source":{"id":"2412.07762","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-10T18:57:12Z","cross_cats_sorted":[],"title_canon_sha256":"01f6188ae65f25c9bb7ef8309d5cd352d57c58e56e551cebca37dc39f33ef182","abstract_canon_sha256":"df614d70855dabc961cc5e71d300ea04b6a190951bd245ceb06c56126c238944"},"schema_version":"1.0"},"canonical_sha256":"b962b3903d556d1c510f273ee4d791cd2c3ccea27190b0cdda0e8b2398e67cdb","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:39.858436Z","signature_b64":"HTRYWArw7BGk3IQGSfsDBPYWF1AR9uXJs1h+6CJ6AiukH+iNqc4ADECeYtQBmPqPzGQFHrKltoS3by+8OkzOBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b962b3903d556d1c510f273ee4d791cd2c3ccea27190b0cdda0e8b2398e67cdb","last_reissued_at":"2026-07-05T11:30:39.857905Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:39.857905Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2412.07762","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:30:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"n4ON+5yde1Ywpnl/jnT4jKGLE+ht2LgXvTaYCbdQFOtguWq+XnPoM3nxgi1SJhhHn/opXYVpQ5K/meTZYADABA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T11:16:18.837705Z"},"content_sha256":"e5782aac61915a1e0c0ec8db156bd21aad54808ab75b48bc4b4136884f428481","schema_version":"1.0","event_id":"sha256:e5782aac61915a1e0c0ec8db156bd21aad54808ab75b48bc4b4136884f428481"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:XFRLHEB5KVWRYUIPE47OJV4RZU","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Efficient Online Reinforcement Learning Fine-Tuning Need Not Retain Offline Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andy Peng, Aviral Kumar, Qiyang Li, Sergey Levine, Zhiyuan Zhou","submitted_at":"2024-12-10T18:57:12Z","abstract_excerpt":"The modern paradigm in machine learning involves pre-training on diverse data, followed by task-specific fine-tuning. In reinforcement learning (RL), this translates to learning via offline RL on a diverse historical dataset, followed by rapid online RL fine-tuning using interaction data. Most RL fine-tuning methods require continued training on offline data for stability and performance. However, this is undesirable because training on diverse offline data is slow and expensive for large datasets, and in principle, also limit the performance improvement possible because of constraints or pess"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.07762","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.07762/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:30:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GLzaBoIR+4pCchuSJHjjls46qGNmLIgN8nl24ZMxaBGNwQx0hrTpJmQSR19QO/hZKOHV9hrrZr0NciRWDuyOBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T11:16:18.838300Z"},"content_sha256":"6693579d09922bddb69bd5711d2040bb78377b725f18201bf2864e718767cb4d","schema_version":"1.0","event_id":"sha256:6693579d09922bddb69bd5711d2040bb78377b725f18201bf2864e718767cb4d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XFRLHEB5KVWRYUIPE47OJV4RZU/bundle.json","state_url":"https://pith.science/pith/XFRLHEB5KVWRYUIPE47OJV4RZU/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XFRLHEB5KVWRYUIPE47OJV4RZU/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T11:16:18Z","links":{"resolver":"https://pith.science/pith/XFRLHEB5KVWRYUIPE47OJV4RZU","bundle":"https://pith.science/pith/XFRLHEB5KVWRYUIPE47OJV4RZU/bundle.json","state":"https://pith.science/pith/XFRLHEB5KVWRYUIPE47OJV4RZU/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XFRLHEB5KVWRYUIPE47OJV4RZU/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:XFRLHEB5KVWRYUIPE47OJV4RZU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"df614d70855dabc961cc5e71d300ea04b6a190951bd245ceb06c56126c238944","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-10T18:57:12Z","title_canon_sha256":"01f6188ae65f25c9bb7ef8309d5cd352d57c58e56e551cebca37dc39f33ef182"},"schema_version":"1.0","source":{"id":"2412.07762","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2412.07762","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"arxiv_version","alias_value":"2412.07762v3","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.07762","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"pith_short_12","alias_value":"XFRLHEB5KVWR","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"pith_short_16","alias_value":"XFRLHEB5KVWRYUIP","created_at":"2026-07-05T11:30:39Z"},{"alias_kind":"pith_short_8","alias_value":"XFRLHEB5","created_at":"2026-07-05T11:30:39Z"}],"graph_snapshots":[{"event_id":"sha256:6693579d09922bddb69bd5711d2040bb78377b725f18201bf2864e718767cb4d","target":"graph","created_at":"2026-07-05T11:30:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2412.07762/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The modern paradigm in machine learning involves pre-training on diverse data, followed by task-specific fine-tuning. In reinforcement learning (RL), this translates to learning via offline RL on a diverse historical dataset, followed by rapid online RL fine-tuning using interaction data. Most RL fine-tuning methods require continued training on offline data for stability and performance. However, this is undesirable because training on diverse offline data is slow and expensive for large datasets, and in principle, also limit the performance improvement possible because of constraints or pess","authors_text":"Andy Peng, Aviral Kumar, Qiyang Li, Sergey Levine, Zhiyuan Zhou","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-10T18:57:12Z","title":"Efficient Online Reinforcement Learning Fine-Tuning Need Not Retain Offline Data"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.07762","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:e5782aac61915a1e0c0ec8db156bd21aad54808ab75b48bc4b4136884f428481","target":"record","created_at":"2026-07-05T11:30:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"df614d70855dabc961cc5e71d300ea04b6a190951bd245ceb06c56126c238944","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-10T18:57:12Z","title_canon_sha256":"01f6188ae65f25c9bb7ef8309d5cd352d57c58e56e551cebca37dc39f33ef182"},"schema_version":"1.0","source":{"id":"2412.07762","kind":"arxiv","version":3}},"canonical_sha256":"b962b3903d556d1c510f273ee4d791cd2c3ccea27190b0cdda0e8b2398e67cdb","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b962b3903d556d1c510f273ee4d791cd2c3ccea27190b0cdda0e8b2398e67cdb","first_computed_at":"2026-07-05T11:30:39.857905Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:30:39.857905Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"HTRYWArw7BGk3IQGSfsDBPYWF1AR9uXJs1h+6CJ6AiukH+iNqc4ADECeYtQBmPqPzGQFHrKltoS3by+8OkzOBA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:30:39.858436Z","signed_message":"canonical_sha256_bytes"},"source_id":"2412.07762","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:e5782aac61915a1e0c0ec8db156bd21aad54808ab75b48bc4b4136884f428481","sha256:6693579d09922bddb69bd5711d2040bb78377b725f18201bf2864e718767cb4d"],"state_sha256":"17097f1c078271f11977a6b49d05998654852b04e96401f3b0ff7cff09093186"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"F3dTaDFhOfSVzUVNOdh5hwTdd15MCyut3BAoYeNKtXbFjHsGng+doQ+FaNy333bDpjoSQCK1fRHEmQ0UkApeDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T11:16:18.844459Z","bundle_sha256":"d93f0575355b5b1310ee24549036b343326364320f2820dd006189a3205a4d79"}}