{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GBR3HFYL7CDGNEBTP3OIMEBL7Q","short_pith_number":"pith:GBR3HFYL","schema_version":"1.0","canonical_sha256":"3063b3970bf8866690337edc86102bfc13850ce08e75d5babcef05c21a2704e9","source":{"kind":"arxiv","id":"2505.16856","version":1},"attestation_state":"computed","paper":{"title":"Efficient Online RL Fine Tuning with Offline Pre-trained Policy Only","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Donglin Wang, Jiacheng Liu, Runze Suo, Shangke Lyu, Wei Xiao, Zifeng Zhuang","submitted_at":"2025-05-22T16:14:08Z","abstract_excerpt":"Improving the performance of pre-trained policies through online reinforcement learning (RL) is a critical yet challenging topic. Existing online RL fine-tuning methods require continued training with offline pretrained Q-functions for stability and performance. However, these offline pretrained Q-functions commonly underestimate state-action pairs beyond the offline dataset due to the conservatism in most offline RL methods, which hinders further exploration when transitioning from the offline to the online setting. Additionally, this requirement limits their applicability in scenarios where "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16856","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-22T16:14:08Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"8625c14d911e118ec8f08c63f65e863767639a85f682226a0dd349902845bd81","abstract_canon_sha256":"a01682a67431f0fb0b5da55ef63859f038b124a031ebee62cf3df19210967dc5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:42.254499Z","signature_b64":"mKXhQxhnCiO0aF6VDkZ8EE3rFSjbOqEz/OB6tqft7WnW1ifqSXKVOlYHk65P8x7sQEq6HgKSIwVUQbOqzpzdDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3063b3970bf8866690337edc86102bfc13850ce08e75d5babcef05c21a2704e9","last_reissued_at":"2026-07-05T11:07:42.253892Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:42.253892Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Online RL Fine Tuning with Offline Pre-trained Policy Only","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Donglin Wang, Jiacheng Liu, Runze Suo, Shangke Lyu, Wei Xiao, Zifeng Zhuang","submitted_at":"2025-05-22T16:14:08Z","abstract_excerpt":"Improving the performance of pre-trained policies through online reinforcement learning (RL) is a critical yet challenging topic. Existing online RL fine-tuning methods require continued training with offline pretrained Q-functions for stability and performance. However, these offline pretrained Q-functions commonly underestimate state-action pairs beyond the offline dataset due to the conservatism in most offline RL methods, which hinders further exploration when transitioning from the offline to the online setting. Additionally, this requirement limits their applicability in scenarios where "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16856","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16856/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16856","created_at":"2026-07-05T11:07:42.253958+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16856v1","created_at":"2026-07-05T11:07:42.253958+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16856","created_at":"2026-07-05T11:07:42.253958+00:00"},{"alias_kind":"pith_short_12","alias_value":"GBR3HFYL7CDG","created_at":"2026-07-05T11:07:42.253958+00:00"},{"alias_kind":"pith_short_16","alias_value":"GBR3HFYL7CDGNEBT","created_at":"2026-07-05T11:07:42.253958+00:00"},{"alias_kind":"pith_short_8","alias_value":"GBR3HFYL","created_at":"2026-07-05T11:07:42.253958+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18675","citing_title":"COOPO: Cyclic Offline-Online Policy Optimization Algorithm","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q","json":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q.json","graph_json":"https://pith.science/api/pith-number/GBR3HFYL7CDGNEBTP3OIMEBL7Q/graph.json","events_json":"https://pith.science/api/pith-number/GBR3HFYL7CDGNEBTP3OIMEBL7Q/events.json","paper":"https://pith.science/paper/GBR3HFYL"},"agent_actions":{"view_html":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q","download_json":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q.json","view_paper":"https://pith.science/paper/GBR3HFYL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16856&json=true","fetch_graph":"https://pith.science/api/pith-number/GBR3HFYL7CDGNEBTP3OIMEBL7Q/graph.json","fetch_events":"https://pith.science/api/pith-number/GBR3HFYL7CDGNEBTP3OIMEBL7Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q/action/storage_attestation","attest_author":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q/action/author_attestation","sign_citation":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q/action/citation_signature","submit_replication":"https://pith.science/pith/GBR3HFYL7CDGNEBTP3OIMEBL7Q/action/replication_record"}},"created_at":"2026-07-05T11:07:42.253958+00:00","updated_at":"2026-07-05T11:07:42.253958+00:00"}