{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:RQDDSC4IILDH5AM5XSGEDA4XM2","short_pith_number":"pith:RQDDSC4I","canonical_record":{"source":{"id":"2604.26644","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-29T13:11:39Z","cross_cats_sorted":[],"title_canon_sha256":"d65152754192bc976f612d7dfebc2faf60965815ef4ca15a5cca47c9d34593ea","abstract_canon_sha256":"6673e736503048114c67c591d67f447d0ea9cda6b0372fdd1d1a56ca0ef5fa49"},"schema_version":"1.0"},"canonical_sha256":"8c06390b8842c67e819dbc8c41839766a32fd47634c04f54f76603f7251e5ece","source":{"kind":"arxiv","id":"2604.26644","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.26644","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"arxiv_version","alias_value":"2604.26644v2","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.26644","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"pith_short_12","alias_value":"RQDDSC4IILDH","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"pith_short_16","alias_value":"RQDDSC4IILDH5AM5","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"pith_short_8","alias_value":"RQDDSC4I","created_at":"2026-08-04T02:00:21Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:RQDDSC4IILDH5AM5XSGEDA4XM2","target":"record","payload":{"canonical_record":{"source":{"id":"2604.26644","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-29T13:11:39Z","cross_cats_sorted":[],"title_canon_sha256":"d65152754192bc976f612d7dfebc2faf60965815ef4ca15a5cca47c9d34593ea","abstract_canon_sha256":"6673e736503048114c67c591d67f447d0ea9cda6b0372fdd1d1a56ca0ef5fa49"},"schema_version":"1.0"},"canonical_sha256":"8c06390b8842c67e819dbc8c41839766a32fd47634c04f54f76603f7251e5ece","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T02:00:21.884034Z","signature_b64":"KfIlnDJT0PF2hMZ+kdliwiX3K+czwURzL7B8Qu9PLLkQ0d1YWbewuE+76zOTa8yIQZpaYkO3mzChx94KhNYPDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c06390b8842c67e819dbc8c41839766a32fd47634c04f54f76603f7251e5ece","last_reissued_at":"2026-08-04T02:00:21.882339Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T02:00:21.882339Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.26644","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-04T02:00:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vM+SvTz9iVan+TGc10YhmuD+dmtef5PAgNR+NWCLOgFqOFP78FL/jb8j3tQqrMA8cCv76tyg6qhf2SHJCFlnBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T10:31:01.769835Z"},"content_sha256":"86becf0cf77357500aa8c304102f74b0ac0350ea98aa830db41cf592b17b27c1","schema_version":"1.0","event_id":"sha256:86becf0cf77357500aa8c304102f74b0ac0350ea98aa830db41cf592b17b27c1"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:RQDDSC4IILDH5AM5XSGEDA4XM2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"When to Vote, When to Rewrite: Disagreement-Guided Strategy Routing for Test-Time Scaling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Output disagreement routes test-time scaling between light fixes, voting, and rewriting to raise accuracy and cut costs on math tasks.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dong Li, Jinpeng Li, Junhua Fang, Juntao Li, Min Zhang, Yixin Ji, Yu Luo, Zhimin Lin","submitted_at":"2026-04-29T13:11:39Z","abstract_excerpt":"Large Reasoning Models (LRMs) achieve strong performance on mathematical reasoning tasks but remain unreliable on challenging instances. Existing test-time scaling methods, such as repeated sampling, self-correction, and tree search, improve performance at the cost of increased computation, yet often exhibit diminishing returns on hard problems. We observe that output disagreement is strongly correlated with instance difficulty and prediction correctness, providing a useful signal for guiding instance-level strategy selection at test time. Based on this insight, we propose a training-free fram"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Experiments on seven mathematical benchmarks and three models show that our method improves accuracy by 3% - 7% while reducing sampling cost compared to existing approaches.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"Output disagreement is strongly correlated with instance difficulty and prediction correctness, providing a reliable signal for guiding instance-level strategy selection at test time.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"A disagreement-guided routing framework dynamically selects among resolution, voting, and rewriting strategies for test-time scaling, delivering 3-7% accuracy gains with lower sampling cost on mathematical benchmarks.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Output disagreement routes test-time scaling between light fixes, voting, and rewriting to raise accuracy and cut costs on math tasks.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"ff563efa07f918c3f0eef34f66b03692efbb351f324ef9b35c964542dfe338fe"},"source":{"id":"2604.26644","kind":"arxiv","version":2},"verdict":{"id":"b546f53a-19d3-4fab-97be-f039980b80ff","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-07T10:55:04.877182Z","strongest_claim":"Experiments on seven mathematical benchmarks and three models show that our method improves accuracy by 3% - 7% while reducing sampling cost compared to existing approaches.","one_line_summary":"A disagreement-guided routing framework dynamically selects among resolution, voting, and rewriting strategies for test-time scaling, delivering 3-7% accuracy gains with lower sampling cost on mathematical benchmarks.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"Output disagreement is strongly correlated with instance difficulty and prediction correctness, providing a reliable signal for guiding instance-level strategy selection at test time.","pith_extraction_headline":"Output disagreement routes test-time scaling between light fixes, voting, and rewriting to raise accuracy and cut costs on math tasks."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.26644/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-20T23:45:34.909834Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T19:57:53.751685Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"f0b23fb2fecc0f3deeffc5dc62fd05043d6bd237b5ff71a2a2c980fe1d5c60f8"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"b546f53a-19d3-4fab-97be-f039980b80ff"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-04T02:00:21Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+KzGnuKWOmAheLxy7FRY+phKjnXkS23IznXhbEU3qynBigcwg2sByDaHSy3TnuorATOLZ+fxhjcaF8TpBRbjAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T10:31:01.770568Z"},"content_sha256":"545e0cbfe00aa6183ffa9129c0b463670d4fca92f3b7691e0ad139a7cd501b66","schema_version":"1.0","event_id":"sha256:545e0cbfe00aa6183ffa9129c0b463670d4fca92f3b7691e0ad139a7cd501b66"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RQDDSC4IILDH5AM5XSGEDA4XM2/bundle.json","state_url":"https://pith.science/pith/RQDDSC4IILDH5AM5XSGEDA4XM2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RQDDSC4IILDH5AM5XSGEDA4XM2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T10:31:01Z","links":{"resolver":"https://pith.science/pith/RQDDSC4IILDH5AM5XSGEDA4XM2","bundle":"https://pith.science/pith/RQDDSC4IILDH5AM5XSGEDA4XM2/bundle.json","state":"https://pith.science/pith/RQDDSC4IILDH5AM5XSGEDA4XM2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RQDDSC4IILDH5AM5XSGEDA4XM2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:RQDDSC4IILDH5AM5XSGEDA4XM2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6673e736503048114c67c591d67f447d0ea9cda6b0372fdd1d1a56ca0ef5fa49","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-29T13:11:39Z","title_canon_sha256":"d65152754192bc976f612d7dfebc2faf60965815ef4ca15a5cca47c9d34593ea"},"schema_version":"1.0","source":{"id":"2604.26644","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.26644","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"arxiv_version","alias_value":"2604.26644v2","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.26644","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"pith_short_12","alias_value":"RQDDSC4IILDH","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"pith_short_16","alias_value":"RQDDSC4IILDH5AM5","created_at":"2026-08-04T02:00:21Z"},{"alias_kind":"pith_short_8","alias_value":"RQDDSC4I","created_at":"2026-08-04T02:00:21Z"}],"graph_snapshots":[{"event_id":"sha256:545e0cbfe00aa6183ffa9129c0b463670d4fca92f3b7691e0ad139a7cd501b66","target":"graph","created_at":"2026-08-04T02:00:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Experiments on seven mathematical benchmarks and three models show that our method improves accuracy by 3% - 7% while reducing sampling cost compared to existing approaches."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"Output disagreement is strongly correlated with instance difficulty and prediction correctness, providing a reliable signal for guiding instance-level strategy selection at test time."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"A disagreement-guided routing framework dynamically selects among resolution, voting, and rewriting strategies for test-time scaling, delivering 3-7% accuracy gains with lower sampling cost on mathematical benchmarks."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Output disagreement routes test-time scaling between light fixes, voting, and rewriting to raise accuracy and cut costs on math tasks."}],"snapshot_sha256":"ff563efa07f918c3f0eef34f66b03692efbb351f324ef9b35c964542dfe338fe"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T23:45:34.909834Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T19:57:53.751685Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.26644/integrity.json","findings":[],"snapshot_sha256":"f0b23fb2fecc0f3deeffc5dc62fd05043d6bd237b5ff71a2a2c980fe1d5c60f8","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large Reasoning Models (LRMs) achieve strong performance on mathematical reasoning tasks but remain unreliable on challenging instances. Existing test-time scaling methods, such as repeated sampling, self-correction, and tree search, improve performance at the cost of increased computation, yet often exhibit diminishing returns on hard problems. We observe that output disagreement is strongly correlated with instance difficulty and prediction correctness, providing a useful signal for guiding instance-level strategy selection at test time. Based on this insight, we propose a training-free fram","authors_text":"Dong Li, Jinpeng Li, Junhua Fang, Juntao Li, Min Zhang, Yixin Ji, Yu Luo, Zhimin Lin","cross_cats":[],"headline":"Output disagreement routes test-time scaling between light fixes, voting, and rewriting to raise accuracy and cut costs on math tasks.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-29T13:11:39Z","title":"When to Vote, When to Rewrite: Disagreement-Guided Strategy Routing for Test-Time Scaling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.26644","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-07T10:55:04.877182Z","id":"b546f53a-19d3-4fab-97be-f039980b80ff","model_set":{"reader":"grok-4.3"},"one_line_summary":"A disagreement-guided routing framework dynamically selects among resolution, voting, and rewriting strategies for test-time scaling, delivering 3-7% accuracy gains with lower sampling cost on mathematical benchmarks.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Output disagreement routes test-time scaling between light fixes, voting, and rewriting to raise accuracy and cut costs on math tasks.","strongest_claim":"Experiments on seven mathematical benchmarks and three models show that our method improves accuracy by 3% - 7% while reducing sampling cost compared to existing approaches.","weakest_assumption":"Output disagreement is strongly correlated with instance difficulty and prediction correctness, providing a reliable signal for guiding instance-level strategy selection at test time."}},"verdict_id":"b546f53a-19d3-4fab-97be-f039980b80ff"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:86becf0cf77357500aa8c304102f74b0ac0350ea98aa830db41cf592b17b27c1","target":"record","created_at":"2026-08-04T02:00:21Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6673e736503048114c67c591d67f447d0ea9cda6b0372fdd1d1a56ca0ef5fa49","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-04-29T13:11:39Z","title_canon_sha256":"d65152754192bc976f612d7dfebc2faf60965815ef4ca15a5cca47c9d34593ea"},"schema_version":"1.0","source":{"id":"2604.26644","kind":"arxiv","version":2}},"canonical_sha256":"8c06390b8842c67e819dbc8c41839766a32fd47634c04f54f76603f7251e5ece","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8c06390b8842c67e819dbc8c41839766a32fd47634c04f54f76603f7251e5ece","first_computed_at":"2026-08-04T02:00:21.882339Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-04T02:00:21.882339Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"KfIlnDJT0PF2hMZ+kdliwiX3K+czwURzL7B8Qu9PLLkQ0d1YWbewuE+76zOTa8yIQZpaYkO3mzChx94KhNYPDA==","signature_status":"signed_v1","signed_at":"2026-08-04T02:00:21.884034Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.26644","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:86becf0cf77357500aa8c304102f74b0ac0350ea98aa830db41cf592b17b27c1","sha256:545e0cbfe00aa6183ffa9129c0b463670d4fca92f3b7691e0ad139a7cd501b66"],"state_sha256":"75215113691a7b0cdd930ecefc1f1cdae9e762230c45e67a039306996fd0d400"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FD/NxrK9S6R54zNIvV+2L9ttQbZ7bwoWwOUbMKCD3K1raDR1HF6wjl0pwkOtms76g1hu/QmKh20qGp5/mC/TBw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T10:31:01.774586Z","bundle_sha256":"eef864fe83c0e3c0f150ccad6be1cecacbc4a5f80be83e5409d82968b7383e71"}}