{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FQ54GI46KRCPAIBDFN2WOGE63S","short_pith_number":"pith:FQ54GI46","schema_version":"1.0","canonical_sha256":"2c3bc3239e5444f020232b7567189edcb0d4b6a5836e81de88f2321258d830aa","source":{"kind":"arxiv","id":"2507.22844","version":1},"attestation_state":"computed","paper":{"title":"RLVMR: Reinforcement Learning with Verifiable Meta-Reasoning Rewards for Robust Long-Horizon Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mingxiao Li, Xiaolong Li, Zhaopeng Tu, Zijing Zhang, Ziyang Chen","submitted_at":"2025-07-30T17:00:48Z","abstract_excerpt":"The development of autonomous agents for complex, long-horizon tasks is a central goal in AI. However, dominant training paradigms face a critical limitation: reinforcement learning (RL) methods that optimize solely for final task success often reinforce flawed or inefficient reasoning paths, a problem we term inefficient exploration. This leads to agents that are brittle and fail to generalize, as they learn to find solutions without learning how to reason coherently. To address this, we introduce RLVMR, a novel framework that integrates dense, process-level supervision into end-to-end RL by "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22844","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-30T17:00:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5f8db465a83fe17fdfd3e699e1f95d4cfccd974440ff8dfcbb7c036b563ad271","abstract_canon_sha256":"5b3324ad16bfe9cda5ab81a5090c15f3e53b41cd73161965f126f7b2627e0742"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:50.243160Z","signature_b64":"0ynBW7JvKr2Ttuu1WqpVXRmob0H9qrEqGDCd53A7WAMEsaelIBh5s6STGczGCXKtGeNzlsEQxB9puJoVgDoWDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c3bc3239e5444f020232b7567189edcb0d4b6a5836e81de88f2321258d830aa","last_reissued_at":"2026-07-05T11:45:50.242625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:50.242625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLVMR: Reinforcement Learning with Verifiable Meta-Reasoning Rewards for Robust Long-Horizon Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mingxiao Li, Xiaolong Li, Zhaopeng Tu, Zijing Zhang, Ziyang Chen","submitted_at":"2025-07-30T17:00:48Z","abstract_excerpt":"The development of autonomous agents for complex, long-horizon tasks is a central goal in AI. However, dominant training paradigms face a critical limitation: reinforcement learning (RL) methods that optimize solely for final task success often reinforce flawed or inefficient reasoning paths, a problem we term inefficient exploration. This leads to agents that are brittle and fail to generalize, as they learn to find solutions without learning how to reason coherently. To address this, we introduce RLVMR, a novel framework that integrates dense, process-level supervision into end-to-end RL by "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22844","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22844","created_at":"2026-07-05T11:45:50.242675+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22844v1","created_at":"2026-07-05T11:45:50.242675+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22844","created_at":"2026-07-05T11:45:50.242675+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQ54GI46KRCP","created_at":"2026-07-05T11:45:50.242675+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQ54GI46KRCPAIBD","created_at":"2026-07-05T11:45:50.242675+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQ54GI46","created_at":"2026-07-05T11:45:50.242675+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25852","citing_title":"Semantic Consistency Policy Optimization for Reinforcement Learning of LLM Agents","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26935","citing_title":"Where Do CoT Training Gains Land in LLM based Agents?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26918","citing_title":"Diagnosing Task Insensitivity in Language Agents","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18089","citing_title":"From Reasoning Traces to Reusable Modules: Understanding Compositional Generalization in Language Model Reasoning","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23989","citing_title":"Towards trustworthy agentic AI: a comprehensive survey of safety, robustness, privacy, and system security","ref_index":206,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21951","citing_title":"Dynamic Mixture of Latent Memories for Self-Evolving Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":269,"is_internal_anchor":false},{"citing_arxiv_id":"2601.12538","citing_title":"Agentic Reasoning for Large Language Models","ref_index":237,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13399","citing_title":"Differentiable Evolutionary Reinforcement Learning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24320","citing_title":"DPEPO: Diverse Parallel Exploration Policy Optimization for LLM-based Agents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07774","citing_title":"RoboAgent: Chaining Basic Capabilities for Embodied Task Planning","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17399","citing_title":"Beyond Meta-Reasoning: Metacognitive Consolidation for Self-Improving LLM Reasoning","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S","json":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S.json","graph_json":"https://pith.science/api/pith-number/FQ54GI46KRCPAIBDFN2WOGE63S/graph.json","events_json":"https://pith.science/api/pith-number/FQ54GI46KRCPAIBDFN2WOGE63S/events.json","paper":"https://pith.science/paper/FQ54GI46"},"agent_actions":{"view_html":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S","download_json":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S.json","view_paper":"https://pith.science/paper/FQ54GI46","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22844&json=true","fetch_graph":"https://pith.science/api/pith-number/FQ54GI46KRCPAIBDFN2WOGE63S/graph.json","fetch_events":"https://pith.science/api/pith-number/FQ54GI46KRCPAIBDFN2WOGE63S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/action/storage_attestation","attest_author":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/action/author_attestation","sign_citation":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/action/citation_signature","submit_replication":"https://pith.science/pith/FQ54GI46KRCPAIBDFN2WOGE63S/action/replication_record"}},"created_at":"2026-07-05T11:45:50.242675+00:00","updated_at":"2026-07-05T11:45:50.242675+00:00"}