{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IHVWNFHOIXTTKNZTEJPQ2GRYVJ","short_pith_number":"pith:IHVWNFHO","schema_version":"1.0","canonical_sha256":"41eb6694ee45e7353733225f0d1a38aa69af8de19f2ec3bb63176eaa646badc4","source":{"kind":"arxiv","id":"2505.14625","version":2},"attestation_state":"computed","paper":{"title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bhaskar Ramasubramanian, Bill Yuchen Lin, Fengqing Jiang, Luyao Niu, Radha Poovendran, Yuetai Li, Zhangchen Xu","submitted_at":"2025-05-20T17:16:44Z","abstract_excerpt":"Reinforcement Learning (RL) has become a powerful tool for enhancing the reasoning abilities of large language models (LLMs) by optimizing their policies with reward signals. Yet, RL's success relies on the reliability of rewards, which are provided by verifiers. In this paper, we expose and analyze a widespread problem--false negatives--where verifiers wrongly reject correct model outputs. Our in-depth study of the Big-Math-RL-Verified dataset reveals that over 38% of model-generated responses suffer from false negatives, where the verifier fails to recognize correct answers. We show, both em"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14625","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-20T17:16:44Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cc3de93324b2cd6e74f2a7b63959799556d6cc4c6ee485f2c332b43b8404de9f","abstract_canon_sha256":"b47c55c1039f39ad3f6b35eea5baabbec760ff153e73cc7ecbe99eacb6bef1e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:25.347633Z","signature_b64":"y3nCS93TT0/w3Lum1Abxo6aLH9OFNlo27zaDZK7ulZcqzZUAYvqQXm1dKw7jRH6KgsC/hzpoQiG5kYGZShPQCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"41eb6694ee45e7353733225f0d1a38aa69af8de19f2ec3bb63176eaa646badc4","last_reissued_at":"2026-07-05T11:07:25.347156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:25.347156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bhaskar Ramasubramanian, Bill Yuchen Lin, Fengqing Jiang, Luyao Niu, Radha Poovendran, Yuetai Li, Zhangchen Xu","submitted_at":"2025-05-20T17:16:44Z","abstract_excerpt":"Reinforcement Learning (RL) has become a powerful tool for enhancing the reasoning abilities of large language models (LLMs) by optimizing their policies with reward signals. Yet, RL's success relies on the reliability of rewards, which are provided by verifiers. In this paper, we expose and analyze a widespread problem--false negatives--where verifiers wrongly reject correct model outputs. Our in-depth study of the Big-Math-RL-Verified dataset reveals that over 38% of model-generated responses suffer from false negatives, where the verifier fails to recognize correct answers. We show, both em"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14625","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14625/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14625","created_at":"2026-07-05T11:07:25.347207+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14625v2","created_at":"2026-07-05T11:07:25.347207+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14625","created_at":"2026-07-05T11:07:25.347207+00:00"},{"alias_kind":"pith_short_12","alias_value":"IHVWNFHOIXTT","created_at":"2026-07-05T11:07:25.347207+00:00"},{"alias_kind":"pith_short_16","alias_value":"IHVWNFHOIXTTKNZT","created_at":"2026-07-05T11:07:25.347207+00:00"},{"alias_kind":"pith_short_8","alias_value":"IHVWNFHO","created_at":"2026-07-05T11:07:25.347207+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03800","citing_title":"Trading Human Curation for Synthetic Augmentation in RLVR","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00915","citing_title":"Reinforcement Learning with Verifiable yet Noisy Rewards under Imperfect Verifiers","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02909","citing_title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ","json":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ.json","graph_json":"https://pith.science/api/pith-number/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/graph.json","events_json":"https://pith.science/api/pith-number/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/events.json","paper":"https://pith.science/paper/IHVWNFHO"},"agent_actions":{"view_html":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ","download_json":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ.json","view_paper":"https://pith.science/paper/IHVWNFHO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14625&json=true","fetch_graph":"https://pith.science/api/pith-number/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/graph.json","fetch_events":"https://pith.science/api/pith-number/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/action/storage_attestation","attest_author":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/action/author_attestation","sign_citation":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/action/citation_signature","submit_replication":"https://pith.science/pith/IHVWNFHOIXTTKNZTEJPQ2GRYVJ/action/replication_record"}},"created_at":"2026-07-05T11:07:25.347207+00:00","updated_at":"2026-07-05T11:07:25.347207+00:00"}