{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:FQ54GI46KRCPAIBDFN2WOGE63S","short_pith_number":"pith:FQ54GI46","canonical_record":{"source":{"id":"2507.22844","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-30T17:00:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5f8db465a83fe17fdfd3e699e1f95d4cfccd974440ff8dfcbb7c036b563ad271","abstract_canon_sha256":"5b3324ad16bfe9cda5ab81a5090c15f3e53b41cd73161965f126f7b2627e0742"},"schema_version":"1.0"},"canonical_sha256":"2c3bc3239e5444f020232b7567189edcb0d4b6a5836e81de88f2321258d830aa","source":{"kind":"arxiv","id":"2507.22844","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.22844","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"arxiv_version","alias_value":"2507.22844v1","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22844","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"pith_short_12","alias_value":"FQ54GI46KRCP","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"pith_short_16","alias_value":"FQ54GI46KRCPAIBD","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"pith_short_8","alias_value":"FQ54GI46","created_at":"2026-07-05T11:45:50Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:FQ54GI46KRCPAIBDFN2WOGE63S","target":"record","payload":{"canonical_record":{"source":{"id":"2507.22844","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-30T17:00:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5f8db465a83fe17fdfd3e699e1f95d4cfccd974440ff8dfcbb7c036b563ad271","abstract_canon_sha256":"5b3324ad16bfe9cda5ab81a5090c15f3e53b41cd73161965f126f7b2627e0742"},"schema_version":"1.0"},"canonical_sha256":"2c3bc3239e5444f020232b7567189edcb0d4b6a5836e81de88f2321258d830aa","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:50.243160Z","signature_b64":"0ynBW7JvKr2Ttuu1WqpVXRmob0H9qrEqGDCd53A7WAMEsaelIBh5s6STGczGCXKtGeNzlsEQxB9puJoVgDoWDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c3bc3239e5444f020232b7567189edcb0d4b6a5836e81de88f2321258d830aa","last_reissued_at":"2026-07-05T11:45:50.242625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:50.242625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2507.22844","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:45:50Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Dn2X2hL4VxcgVN4QbbP5Euvk6MQFYgXfnjQwWI65NVjEQ69X3kjOs1XxEpvsl1k0ZssQsj5AyFDLxGypsGcfBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T17:03:57.166739Z"},"content_sha256":"b938d36e600017300517ebf25c05cf60372be1a73c0b8b88d486ff67c35c08f5","schema_version":"1.0","event_id":"sha256:b938d36e600017300517ebf25c05cf60372be1a73c0b8b88d486ff67c35c08f5"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:FQ54GI46KRCPAIBDFN2WOGE63S","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"RLVMR: Reinforcement Learning with Verifiable Meta-Reasoning Rewards for Robust Long-Horizon Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mingxiao Li, Xiaolong Li, Zhaopeng Tu, Zijing Zhang, Ziyang Chen","submitted_at":"2025-07-30T17:00:48Z","abstract_excerpt":"The development of autonomous agents for complex, long-horizon tasks is a central goal in AI. However, dominant training paradigms face a critical limitation: reinforcement learning (RL) methods that optimize solely for final task success often reinforce flawed or inefficient reasoning paths, a problem we term inefficient exploration. This leads to agents that are brittle and fail to generalize, as they learn to find solutions without learning how to reason coherently. To address this, we introduce RLVMR, a novel framework that integrates dense, process-level supervision into end-to-end RL by "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22844","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:45:50Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5R5VOvH2Xe/gQ4WTpPqGwbbbwBmLbCki54LUnH4ISNubZCrr8TMhjrRvPqcH9m52qRAQftHwVWfa+nTDZcZGDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T17:03:57.167192Z"},"content_sha256":"d93e9fa9f685caf502ae2773c92aa94a3b69c90391bb3cf8bdc0e1de05b002d3","schema_version":"1.0","event_id":"sha256:d93e9fa9f685caf502ae2773c92aa94a3b69c90391bb3cf8bdc0e1de05b002d3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/bundle.json","state_url":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FQ54GI46KRCPAIBDFN2WOGE63S/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T17:03:57Z","links":{"resolver":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S","bundle":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/bundle.json","state":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FQ54GI46KRCPAIBDFN2WOGE63S/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:FQ54GI46KRCPAIBDFN2WOGE63S","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5b3324ad16bfe9cda5ab81a5090c15f3e53b41cd73161965f126f7b2627e0742","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-30T17:00:48Z","title_canon_sha256":"5f8db465a83fe17fdfd3e699e1f95d4cfccd974440ff8dfcbb7c036b563ad271"},"schema_version":"1.0","source":{"id":"2507.22844","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.22844","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"arxiv_version","alias_value":"2507.22844v1","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22844","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"pith_short_12","alias_value":"FQ54GI46KRCP","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"pith_short_16","alias_value":"FQ54GI46KRCPAIBD","created_at":"2026-07-05T11:45:50Z"},{"alias_kind":"pith_short_8","alias_value":"FQ54GI46","created_at":"2026-07-05T11:45:50Z"}],"graph_snapshots":[{"event_id":"sha256:d93e9fa9f685caf502ae2773c92aa94a3b69c90391bb3cf8bdc0e1de05b002d3","target":"graph","created_at":"2026-07-05T11:45:50Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2507.22844/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The development of autonomous agents for complex, long-horizon tasks is a central goal in AI. However, dominant training paradigms face a critical limitation: reinforcement learning (RL) methods that optimize solely for final task success often reinforce flawed or inefficient reasoning paths, a problem we term inefficient exploration. This leads to agents that are brittle and fail to generalize, as they learn to find solutions without learning how to reason coherently. To address this, we introduce RLVMR, a novel framework that integrates dense, process-level supervision into end-to-end RL by ","authors_text":"Mingxiao Li, Xiaolong Li, Zhaopeng Tu, Zijing Zhang, Ziyang Chen","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-30T17:00:48Z","title":"RLVMR: Reinforcement Learning with Verifiable Meta-Reasoning Rewards for Robust Long-Horizon Agents"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22844","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b938d36e600017300517ebf25c05cf60372be1a73c0b8b88d486ff67c35c08f5","target":"record","created_at":"2026-07-05T11:45:50Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5b3324ad16bfe9cda5ab81a5090c15f3e53b41cd73161965f126f7b2627e0742","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-30T17:00:48Z","title_canon_sha256":"5f8db465a83fe17fdfd3e699e1f95d4cfccd974440ff8dfcbb7c036b563ad271"},"schema_version":"1.0","source":{"id":"2507.22844","kind":"arxiv","version":1}},"canonical_sha256":"2c3bc3239e5444f020232b7567189edcb0d4b6a5836e81de88f2321258d830aa","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2c3bc3239e5444f020232b7567189edcb0d4b6a5836e81de88f2321258d830aa","first_computed_at":"2026-07-05T11:45:50.242625Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:45:50.242625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"0ynBW7JvKr2Ttuu1WqpVXRmob0H9qrEqGDCd53A7WAMEsaelIBh5s6STGczGCXKtGeNzlsEQxB9puJoVgDoWDg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:45:50.243160Z","signed_message":"canonical_sha256_bytes"},"source_id":"2507.22844","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b938d36e600017300517ebf25c05cf60372be1a73c0b8b88d486ff67c35c08f5","sha256:d93e9fa9f685caf502ae2773c92aa94a3b69c90391bb3cf8bdc0e1de05b002d3"],"state_sha256":"6793ec4a14c10fd65bd8cbe86f9c8f5775788a8f52c5fb72b1119cb689327d0b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/73vkr3djbhXQN4NK9BTlkkv8h/vjVDtKF7AAeixwQxHAF2RzkwVPB7GDII97ebpMKwiTzPwoDTY+brpIlT/AQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T17:03:57.170140Z","bundle_sha256":"efc81e6b79673f3b1026c3d8e0772b689c95d4f2d946c68e3c998036e8e33e7e"}}