{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:GIHBBBL62YK3SEKE7QNB77WE4J","short_pith_number":"pith:GIHBBBL6","schema_version":"1.0","canonical_sha256":"320e10857ed615b91144fc1a1ffec4e259812e3abe671f96b10820ec29280616","source":{"kind":"arxiv","id":"2607.04364","version":1},"attestation_state":"computed","paper":{"title":"RL Forgets! Towards Continual Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Ye, Jian Zhao, Mao-Lin Luo, Min-Ling Zhang, Tong Wei, Zhe-Xu Wang, Zi-Hao Zhou","submitted_at":"2026-07-05T15:46:11Z","abstract_excerpt":"Continual post-training is becoming a central paradigm for adapting vision-language models to evolving tasks. Recent work has increasingly favored reinforcement learning over supervised fine-tuning, driven by the belief that reinforcement learning is inherently less prone to forgetting. However, the belief remains insufficiently validated, as existing evidence is largely drawn from outdated or homogeneous benchmarks. To revisit this assumption, we introduce MRCL, a Multimodal Reasoning Continual Learning benchmark built from diverse and recently released multimodal datasets. Experiments on MRC"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.04364","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-05T15:46:11Z","cross_cats_sorted":[],"title_canon_sha256":"52300d345189434432a02829952de626df8d26975c35c64806a29cc97fa6efd2","abstract_canon_sha256":"42c3acb3193356b4e5d787306631ef860a6e9c16766d2af049ca6fa6b1967ebb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:19:10.922350Z","signature_b64":"b4Y1ZxPtOkH7+vqodMxcS/UlCNA+QXj6Ifj3lS24Oaut6aXVkILkHisIAJqai9TPXvNl7lVqXdNpDcxmR49dCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"320e10857ed615b91144fc1a1ffec4e259812e3abe671f96b10820ec29280616","last_reissued_at":"2026-07-07T02:19:10.921625Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:19:10.921625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RL Forgets! Towards Continual Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Ye, Jian Zhao, Mao-Lin Luo, Min-Ling Zhang, Tong Wei, Zhe-Xu Wang, Zi-Hao Zhou","submitted_at":"2026-07-05T15:46:11Z","abstract_excerpt":"Continual post-training is becoming a central paradigm for adapting vision-language models to evolving tasks. Recent work has increasingly favored reinforcement learning over supervised fine-tuning, driven by the belief that reinforcement learning is inherently less prone to forgetting. However, the belief remains insufficiently validated, as existing evidence is largely drawn from outdated or homogeneous benchmarks. To revisit this assumption, we introduce MRCL, a Multimodal Reasoning Continual Learning benchmark built from diverse and recently released multimodal datasets. Experiments on MRC"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.04364","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.04364/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.04364","created_at":"2026-07-07T02:19:10.921740+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.04364v1","created_at":"2026-07-07T02:19:10.921740+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.04364","created_at":"2026-07-07T02:19:10.921740+00:00"},{"alias_kind":"pith_short_12","alias_value":"GIHBBBL62YK3","created_at":"2026-07-07T02:19:10.921740+00:00"},{"alias_kind":"pith_short_16","alias_value":"GIHBBBL62YK3SEKE","created_at":"2026-07-07T02:19:10.921740+00:00"},{"alias_kind":"pith_short_8","alias_value":"GIHBBBL6","created_at":"2026-07-07T02:19:10.921740+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J","json":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J.json","graph_json":"https://pith.science/api/pith-number/GIHBBBL62YK3SEKE7QNB77WE4J/graph.json","events_json":"https://pith.science/api/pith-number/GIHBBBL62YK3SEKE7QNB77WE4J/events.json","paper":"https://pith.science/paper/GIHBBBL6"},"agent_actions":{"view_html":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J","download_json":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J.json","view_paper":"https://pith.science/paper/GIHBBBL6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.04364&json=true","fetch_graph":"https://pith.science/api/pith-number/GIHBBBL62YK3SEKE7QNB77WE4J/graph.json","fetch_events":"https://pith.science/api/pith-number/GIHBBBL62YK3SEKE7QNB77WE4J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J/action/storage_attestation","attest_author":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J/action/author_attestation","sign_citation":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J/action/citation_signature","submit_replication":"https://pith.science/pith/GIHBBBL62YK3SEKE7QNB77WE4J/action/replication_record"}},"created_at":"2026-07-07T02:19:10.921740+00:00","updated_at":"2026-07-07T02:19:10.921740+00:00"}