{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:WL63WTSC3M7RLC22TVPRCYJX26","short_pith_number":"pith:WL63WTSC","canonical_record":{"source":{"id":"2509.01321","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-01T10:04:20Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e47e34f316a91e566d1e80ee06aefd030b08340e0aa4bf0fbff3bb9796e98048","abstract_canon_sha256":"669017d95238a5860de849bb3fbf92c93c5744fb9d60f970205371efb2a207f4"},"schema_version":"1.0"},"canonical_sha256":"b2fdbb4e42db3f158b5a9d5f116137d7ab1f91a1b28a5fbc7092f3216f407f49","source":{"kind":"arxiv","id":"2509.01321","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.01321","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"arxiv_version","alias_value":"2509.01321v1","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.01321","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"pith_short_12","alias_value":"WL63WTSC3M7R","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"pith_short_16","alias_value":"WL63WTSC3M7RLC22","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"pith_short_8","alias_value":"WL63WTSC","created_at":"2026-07-05T12:02:58Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:WL63WTSC3M7RLC22TVPRCYJX26","target":"record","payload":{"canonical_record":{"source":{"id":"2509.01321","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-01T10:04:20Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e47e34f316a91e566d1e80ee06aefd030b08340e0aa4bf0fbff3bb9796e98048","abstract_canon_sha256":"669017d95238a5860de849bb3fbf92c93c5744fb9d60f970205371efb2a207f4"},"schema_version":"1.0"},"canonical_sha256":"b2fdbb4e42db3f158b5a9d5f116137d7ab1f91a1b28a5fbc7092f3216f407f49","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:58.223365Z","signature_b64":"8pq8/LjyL8FJjs6cic1lwcH9wtYI7imFBEorYPMAGNSsOWsqUZB+KmipZsu+/pqs00WFZU/oKcX7WQL7LSjmAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b2fdbb4e42db3f158b5a9d5f116137d7ab1f91a1b28a5fbc7092f3216f407f49","last_reissued_at":"2026-07-05T12:02:58.222805Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:58.222805Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2509.01321","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:02:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"r2ORAclMB4zDTnhJeXIpaS5jdfWJMRbwt+c1jD6QblqxwanDNTRb5fWwVd6z1+8MP6EsyVUzQL01ZAC+1siGDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T19:52:15.582506Z"},"content_sha256":"3229f01e82423751540a08b7b9a98867493546e85be1dbce2a685ddc2d56801c","schema_version":"1.0","event_id":"sha256:3229f01e82423751540a08b7b9a98867493546e85be1dbce2a685ddc2d56801c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:WL63WTSC3M7RLC22TVPRCYJX26","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Towards High Data Efficiency in Reinforcement Learning with Verifiable Reward","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jun Zhou, Wayne Xin Zhao, Xinyu Tang, Yurou Liu, Zhenduo Zhang, Zhiqiang Zhang, Zujie Wen","submitted_at":"2025-09-01T10:04:20Z","abstract_excerpt":"Recent advances in large reasoning models have leveraged reinforcement learning with verifiable rewards (RLVR) to improve reasoning capabilities. However, scaling these methods typically requires extensive rollout computation and large datasets, leading to high training costs and low data efficiency. To mitigate this issue, we propose DEPO, a Data-Efficient Policy Optimization pipeline that combines optimized strategies for both offline and online data selection. In the offline phase, we curate a high-quality subset of training samples based on diversity, influence, and appropriate difficulty."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.01321","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.01321/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:02:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FTvMXKL1TVcqWcIgSF3uBYMU7pAjOcJGJF2Alh4WAHu1yrhelft5Xr2ZoQMUG2KaybfDD65CfuzuP0g+HdBvCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T19:52:15.583836Z"},"content_sha256":"aafd6549d2826b1898cf2acefd187ba4ec004a8a88fa08ac8a8fe6d2e8b44ec8","schema_version":"1.0","event_id":"sha256:aafd6549d2826b1898cf2acefd187ba4ec004a8a88fa08ac8a8fe6d2e8b44ec8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WL63WTSC3M7RLC22TVPRCYJX26/bundle.json","state_url":"https://pith.science/pith/WL63WTSC3M7RLC22TVPRCYJX26/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WL63WTSC3M7RLC22TVPRCYJX26/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T19:52:15Z","links":{"resolver":"https://pith.science/pith/WL63WTSC3M7RLC22TVPRCYJX26","bundle":"https://pith.science/pith/WL63WTSC3M7RLC22TVPRCYJX26/bundle.json","state":"https://pith.science/pith/WL63WTSC3M7RLC22TVPRCYJX26/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WL63WTSC3M7RLC22TVPRCYJX26/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:WL63WTSC3M7RLC22TVPRCYJX26","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"669017d95238a5860de849bb3fbf92c93c5744fb9d60f970205371efb2a207f4","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-01T10:04:20Z","title_canon_sha256":"e47e34f316a91e566d1e80ee06aefd030b08340e0aa4bf0fbff3bb9796e98048"},"schema_version":"1.0","source":{"id":"2509.01321","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.01321","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"arxiv_version","alias_value":"2509.01321v1","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.01321","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"pith_short_12","alias_value":"WL63WTSC3M7R","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"pith_short_16","alias_value":"WL63WTSC3M7RLC22","created_at":"2026-07-05T12:02:58Z"},{"alias_kind":"pith_short_8","alias_value":"WL63WTSC","created_at":"2026-07-05T12:02:58Z"}],"graph_snapshots":[{"event_id":"sha256:aafd6549d2826b1898cf2acefd187ba4ec004a8a88fa08ac8a8fe6d2e8b44ec8","target":"graph","created_at":"2026-07-05T12:02:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2509.01321/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advances in large reasoning models have leveraged reinforcement learning with verifiable rewards (RLVR) to improve reasoning capabilities. However, scaling these methods typically requires extensive rollout computation and large datasets, leading to high training costs and low data efficiency. To mitigate this issue, we propose DEPO, a Data-Efficient Policy Optimization pipeline that combines optimized strategies for both offline and online data selection. In the offline phase, we curate a high-quality subset of training samples based on diversity, influence, and appropriate difficulty.","authors_text":"Jun Zhou, Wayne Xin Zhao, Xinyu Tang, Yurou Liu, Zhenduo Zhang, Zhiqiang Zhang, Zujie Wen","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-01T10:04:20Z","title":"Towards High Data Efficiency in Reinforcement Learning with Verifiable Reward"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.01321","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3229f01e82423751540a08b7b9a98867493546e85be1dbce2a685ddc2d56801c","target":"record","created_at":"2026-07-05T12:02:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"669017d95238a5860de849bb3fbf92c93c5744fb9d60f970205371efb2a207f4","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-01T10:04:20Z","title_canon_sha256":"e47e34f316a91e566d1e80ee06aefd030b08340e0aa4bf0fbff3bb9796e98048"},"schema_version":"1.0","source":{"id":"2509.01321","kind":"arxiv","version":1}},"canonical_sha256":"b2fdbb4e42db3f158b5a9d5f116137d7ab1f91a1b28a5fbc7092f3216f407f49","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b2fdbb4e42db3f158b5a9d5f116137d7ab1f91a1b28a5fbc7092f3216f407f49","first_computed_at":"2026-07-05T12:02:58.222805Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:02:58.222805Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"8pq8/LjyL8FJjs6cic1lwcH9wtYI7imFBEorYPMAGNSsOWsqUZB+KmipZsu+/pqs00WFZU/oKcX7WQL7LSjmAA==","signature_status":"signed_v1","signed_at":"2026-07-05T12:02:58.223365Z","signed_message":"canonical_sha256_bytes"},"source_id":"2509.01321","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3229f01e82423751540a08b7b9a98867493546e85be1dbce2a685ddc2d56801c","sha256:aafd6549d2826b1898cf2acefd187ba4ec004a8a88fa08ac8a8fe6d2e8b44ec8"],"state_sha256":"7b90dfbc3636c717ce0b937be5acc142d171f1b7856a59caec04055c683924b1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"pm6AaycvVvthk93tKm2bvkb38r2kM37fqX++Cj9RhfYoV5VIygXMVz0z2S89oR11UamnWtPk6TvSXwaSD6piAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T19:52:15.594088Z","bundle_sha256":"462678eb82af617ae6458140584287bf9ad9cce3a698c74f1f3638ec5ea8085c"}}