{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:KRCLUOGPAZHHUX6VZGUBMEC4QM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d9ece7e17c91cc048adcb9588b17b78a604666326b537951f7e21c488e611c9f","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-13T02:46:53Z","title_canon_sha256":"d7e4c22b203c6493d6a22a35a4a4631d712370862e786552f346295106fd357c"},"schema_version":"1.0","source":{"id":"2506.11425","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.11425","created_at":"2026-07-05T11:25:12Z"},{"alias_kind":"arxiv_version","alias_value":"2506.11425v2","created_at":"2026-07-05T11:25:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11425","created_at":"2026-07-05T11:25:12Z"},{"alias_kind":"pith_short_12","alias_value":"KRCLUOGPAZHH","created_at":"2026-07-05T11:25:12Z"},{"alias_kind":"pith_short_16","alias_value":"KRCLUOGPAZHHUX6V","created_at":"2026-07-05T11:25:12Z"},{"alias_kind":"pith_short_8","alias_value":"KRCLUOGP","created_at":"2026-07-05T11:25:12Z"}],"graph_snapshots":[{"event_id":"sha256:130f5e04f9049360e74c9792a3153242cf79ef78db6ea5168070b43c143ad5a5","target":"graph","created_at":"2026-07-05T11:25:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.11425/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Verifiable Rewards (RLVR) has been widely adopted as the de facto method for enhancing the reasoning capabilities of large language models and has demonstrated notable success in verifiable domains like math and competitive programming tasks. However, the efficacy of RLVR diminishes significantly when applied to agentic environments. These settings, characterized by multi-step, complex problem solving, lead to high failure rates even for frontier LLMs, as the reward landscape is too sparse for effective model training via conventional RLVR. In this work, we introduc","authors_text":"Clinton Wang, Jeff Da, Nikhil Barhate, Sean Hendryx, Xiang Deng, Yuntao Ma","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-13T02:46:53Z","title":"Agent-RLVR: Training Software Engineering Agents via Guidance and Environment Rewards"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11425","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f545002abfe2ef8000a38e5ca14d4dabccb0c101959815b5aab7937e4b958613","target":"record","created_at":"2026-07-05T11:25:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d9ece7e17c91cc048adcb9588b17b78a604666326b537951f7e21c488e611c9f","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-13T02:46:53Z","title_canon_sha256":"d7e4c22b203c6493d6a22a35a4a4631d712370862e786552f346295106fd357c"},"schema_version":"1.0","source":{"id":"2506.11425","kind":"arxiv","version":2}},"canonical_sha256":"5444ba38cf064e7a5fd5c9a816105c8336ca428208545b6fea053f31fcd4f9e7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5444ba38cf064e7a5fd5c9a816105c8336ca428208545b6fea053f31fcd4f9e7","first_computed_at":"2026-07-05T11:25:12.451173Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:25:12.451173Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"KWDhGSBHeYVpM1imNEvQOXElvatHuoZRTmcU4Xk1qfd1pdtzKkx2LZTgYh2Sa7k4DWBZv0PXyuonVCD+aeNPBw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:25:12.451668Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.11425","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f545002abfe2ef8000a38e5ca14d4dabccb0c101959815b5aab7937e4b958613","sha256:130f5e04f9049360e74c9792a3153242cf79ef78db6ea5168070b43c143ad5a5"],"state_sha256":"289d673b2775b963a4ffd77f1b66baaea25e1fc9267e8b67c424cabfe6abaa43"}