{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KRCLUOGPAZHHUX6VZGUBMEC4QM","short_pith_number":"pith:KRCLUOGP","schema_version":"1.0","canonical_sha256":"5444ba38cf064e7a5fd5c9a816105c8336ca428208545b6fea053f31fcd4f9e7","source":{"kind":"arxiv","id":"2506.11425","version":2},"attestation_state":"computed","paper":{"title":"Agent-RLVR: Training Software Engineering Agents via Guidance and Environment Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Clinton Wang, Jeff Da, Nikhil Barhate, Sean Hendryx, Xiang Deng, Yuntao Ma","submitted_at":"2025-06-13T02:46:53Z","abstract_excerpt":"Reinforcement Learning from Verifiable Rewards (RLVR) has been widely adopted as the de facto method for enhancing the reasoning capabilities of large language models and has demonstrated notable success in verifiable domains like math and competitive programming tasks. However, the efficacy of RLVR diminishes significantly when applied to agentic environments. These settings, characterized by multi-step, complex problem solving, lead to high failure rates even for frontier LLMs, as the reward landscape is too sparse for effective model training via conventional RLVR. In this work, we introduc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.11425","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-13T02:46:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d7e4c22b203c6493d6a22a35a4a4631d712370862e786552f346295106fd357c","abstract_canon_sha256":"d9ece7e17c91cc048adcb9588b17b78a604666326b537951f7e21c488e611c9f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:12.451668Z","signature_b64":"KWDhGSBHeYVpM1imNEvQOXElvatHuoZRTmcU4Xk1qfd1pdtzKkx2LZTgYh2Sa7k4DWBZv0PXyuonVCD+aeNPBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5444ba38cf064e7a5fd5c9a816105c8336ca428208545b6fea053f31fcd4f9e7","last_reissued_at":"2026-07-05T11:25:12.451173Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:12.451173Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Agent-RLVR: Training Software Engineering Agents via Guidance and Environment Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Clinton Wang, Jeff Da, Nikhil Barhate, Sean Hendryx, Xiang Deng, Yuntao Ma","submitted_at":"2025-06-13T02:46:53Z","abstract_excerpt":"Reinforcement Learning from Verifiable Rewards (RLVR) has been widely adopted as the de facto method for enhancing the reasoning capabilities of large language models and has demonstrated notable success in verifiable domains like math and competitive programming tasks. However, the efficacy of RLVR diminishes significantly when applied to agentic environments. These settings, characterized by multi-step, complex problem solving, lead to high failure rates even for frontier LLMs, as the reward landscape is too sparse for effective model training via conventional RLVR. In this work, we introduc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11425","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.11425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.11425","created_at":"2026-07-05T11:25:12.451223+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.11425v2","created_at":"2026-07-05T11:25:12.451223+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11425","created_at":"2026-07-05T11:25:12.451223+00:00"},{"alias_kind":"pith_short_12","alias_value":"KRCLUOGPAZHH","created_at":"2026-07-05T11:25:12.451223+00:00"},{"alias_kind":"pith_short_16","alias_value":"KRCLUOGPAZHHUX6V","created_at":"2026-07-05T11:25:12.451223+00:00"},{"alias_kind":"pith_short_8","alias_value":"KRCLUOGP","created_at":"2026-07-05T11:25:12.451223+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01465","citing_title":"Beyond Next-Token Prediction: An RLVR Proof of Concept for Tool-Use Agents on Atlassian Workflows","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12908","citing_title":"SENTINEL: Failure-Driven Reinforcement Learning for Training Tool-Using Language Model Agents","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03963","citing_title":"AgenticRL: Self-Refining Agentic Reinforcement Learning for Vision-Conditioned UAV Navigation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03800","citing_title":"Trading Human Curation for Synthetic Augmentation in RLVR","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01619","citing_title":"ReSkill: Reconciling Skill Creation with Policy Optimization in Agentic RL","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12969","citing_title":"Revisiting Reinforcement Learning with Verifiable Rewards from a Contrastive Perspective","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12969","citing_title":"Revisiting Reinforcement Learning with Verifiable Rewards from a Contrastive Perspective","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16941","citing_title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10493","citing_title":"SWE-Shepherd: Advancing PRMs for Reinforcing Code Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM","json":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM.json","graph_json":"https://pith.science/api/pith-number/KRCLUOGPAZHHUX6VZGUBMEC4QM/graph.json","events_json":"https://pith.science/api/pith-number/KRCLUOGPAZHHUX6VZGUBMEC4QM/events.json","paper":"https://pith.science/paper/KRCLUOGP"},"agent_actions":{"view_html":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM","download_json":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM.json","view_paper":"https://pith.science/paper/KRCLUOGP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.11425&json=true","fetch_graph":"https://pith.science/api/pith-number/KRCLUOGPAZHHUX6VZGUBMEC4QM/graph.json","fetch_events":"https://pith.science/api/pith-number/KRCLUOGPAZHHUX6VZGUBMEC4QM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM/action/storage_attestation","attest_author":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM/action/author_attestation","sign_citation":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM/action/citation_signature","submit_replication":"https://pith.science/pith/KRCLUOGPAZHHUX6VZGUBMEC4QM/action/replication_record"}},"created_at":"2026-07-05T11:25:12.451223+00:00","updated_at":"2026-07-05T11:25:12.451223+00:00"}