{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JJORJ2IN6BOY3DFLB4LRVZ4XRP","short_pith_number":"pith:JJORJ2IN","schema_version":"1.0","canonical_sha256":"4a5d14e90df05d8d8cab0f171ae7978bdfafdb3ae7b62e4f898a69de7f15f409","source":{"kind":"arxiv","id":"2504.10337","version":2},"attestation_state":"computed","paper":{"title":"Heimdall: test-time scaling on the generative verification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Wenlei Shi, Xing Jin","submitted_at":"2025-04-14T15:46:33Z","abstract_excerpt":"An AI system can create and maintain knowledge only to the extent that it can verify that knowledge itself. Recent work on long Chain-of-Thought reasoning has demonstrated great potential of LLMs on solving competitive problems, but their verification ability remains to be weak and not sufficiently investigated. In this paper, we propose Heimdall, the long CoT verification LLM that can accurately judge the correctness of solutions. With pure reinforcement learning, we boost the verification accuracy from 62.5% to 94.5% on competitive math problems. By scaling with repeated sampling, the accura"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.10337","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-04-14T15:46:33Z","cross_cats_sorted":[],"title_canon_sha256":"4c0e7ee4a791603156423264f8b61032b6f29899fb6c06d1f2b0e473f998e011","abstract_canon_sha256":"615b026194c6610e2ae74768b5df5f8810801e43431a4e13f68ff1f685f6869a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:49:53.743257Z","signature_b64":"2dKFLhaWFm7QpHRCqWdT0bqBWKPOTpszM5Ezdz2VOpPsXI0zf6BtgPQj3IPM03M7/psJecNDuU2h5/jDbcNgCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a5d14e90df05d8d8cab0f171ae7978bdfafdb3ae7b62e4f898a69de7f15f409","last_reissued_at":"2026-07-05T10:49:53.742832Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:49:53.742832Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Heimdall: test-time scaling on the generative verification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Wenlei Shi, Xing Jin","submitted_at":"2025-04-14T15:46:33Z","abstract_excerpt":"An AI system can create and maintain knowledge only to the extent that it can verify that knowledge itself. Recent work on long Chain-of-Thought reasoning has demonstrated great potential of LLMs on solving competitive problems, but their verification ability remains to be weak and not sufficiently investigated. In this paper, we propose Heimdall, the long CoT verification LLM that can accurately judge the correctness of solutions. With pure reinforcement learning, we boost the verification accuracy from 62.5% to 94.5% on competitive math problems. By scaling with repeated sampling, the accura"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.10337","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.10337/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.10337","created_at":"2026-07-05T10:49:53.742887+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.10337v2","created_at":"2026-07-05T10:49:53.742887+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.10337","created_at":"2026-07-05T10:49:53.742887+00:00"},{"alias_kind":"pith_short_12","alias_value":"JJORJ2IN6BOY","created_at":"2026-07-05T10:49:53.742887+00:00"},{"alias_kind":"pith_short_16","alias_value":"JJORJ2IN6BOY3DFL","created_at":"2026-07-05T10:49:53.742887+00:00"},{"alias_kind":"pith_short_8","alias_value":"JJORJ2IN","created_at":"2026-07-05T10:49:53.742887+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20531","citing_title":"Pseudo-Formalization for Automatic Proof Verification","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20531","citing_title":"Pseudo-Formalization for Automatic Proof Verification","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19880","citing_title":"What If Consensus Lies? Selective-Complementary Reinforcement Learning at Test Time","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP","json":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP.json","graph_json":"https://pith.science/api/pith-number/JJORJ2IN6BOY3DFLB4LRVZ4XRP/graph.json","events_json":"https://pith.science/api/pith-number/JJORJ2IN6BOY3DFLB4LRVZ4XRP/events.json","paper":"https://pith.science/paper/JJORJ2IN"},"agent_actions":{"view_html":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP","download_json":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP.json","view_paper":"https://pith.science/paper/JJORJ2IN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.10337&json=true","fetch_graph":"https://pith.science/api/pith-number/JJORJ2IN6BOY3DFLB4LRVZ4XRP/graph.json","fetch_events":"https://pith.science/api/pith-number/JJORJ2IN6BOY3DFLB4LRVZ4XRP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP/action/storage_attestation","attest_author":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP/action/author_attestation","sign_citation":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP/action/citation_signature","submit_replication":"https://pith.science/pith/JJORJ2IN6BOY3DFLB4LRVZ4XRP/action/replication_record"}},"created_at":"2026-07-05T10:49:53.742887+00:00","updated_at":"2026-07-05T10:49:53.742887+00:00"}