{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:XBPOS36Y4FV3OH5FAKVBIUS5I4","short_pith_number":"pith:XBPOS36Y","canonical_record":{"source":{"id":"2608.12585","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-12T21:00:22Z","cross_cats_sorted":[],"title_canon_sha256":"8f73098ec08a2aca09d12ed06323a04b128ffb450bbd2d9b66ad27c4752a0148","abstract_canon_sha256":"da8251d17c898d8f4e6b1e9984f93f83792aab5c3953d0071fc52774b95da056"},"schema_version":"1.0"},"canonical_sha256":"b85ee96fd8e16bb71fa502aa14525d4712cc5ef37129c6ba8b5cb4c6783bdb35","source":{"kind":"arxiv","id":"2608.12585","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2608.12585","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"arxiv_version","alias_value":"2608.12585v1","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.12585","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"pith_short_12","alias_value":"XBPOS36Y4FV3","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"pith_short_16","alias_value":"XBPOS36Y4FV3OH5F","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"pith_short_8","alias_value":"XBPOS36Y","created_at":"2026-08-14T00:45:27Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:XBPOS36Y4FV3OH5FAKVBIUS5I4","target":"record","payload":{"canonical_record":{"source":{"id":"2608.12585","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-12T21:00:22Z","cross_cats_sorted":[],"title_canon_sha256":"8f73098ec08a2aca09d12ed06323a04b128ffb450bbd2d9b66ad27c4752a0148","abstract_canon_sha256":"da8251d17c898d8f4e6b1e9984f93f83792aab5c3953d0071fc52774b95da056"},"schema_version":"1.0"},"canonical_sha256":"b85ee96fd8e16bb71fa502aa14525d4712cc5ef37129c6ba8b5cb4c6783bdb35","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-14T00:45:27.951316Z","signature_b64":"jMl57bGlime3HdlHbrX5U8hTwG/3UkQmr+3q9ER+dWO5Sw9DiKnwZvpUPhAWcBF9UT9sufkbWSWJWwRckVQZDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b85ee96fd8e16bb71fa502aa14525d4712cc5ef37129c6ba8b5cb4c6783bdb35","last_reissued_at":"2026-08-14T00:45:27.941117Z","signature_status":"signed_v1","first_computed_at":"2026-08-14T00:45:27.941117Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2608.12585","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-14T00:45:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rA46J/TK0W1klycaheLMsprkvHN98E35/HHjhjQxyTs+YjNnJonteXYi+mZ0ziq2ANt04n+jp/yvzEu1DZiABQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T08:57:17.050331Z"},"content_sha256":"046667464531427a3fa5dbda35745a544997c079f4cf34a84d49727cff65776c","schema_version":"1.0","event_id":"sha256:046667464531427a3fa5dbda35745a544997c079f4cf34a84d49727cff65776c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:XBPOS36Y4FV3OH5FAKVBIUS5I4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reasoning Jury: Multi-Model Consensus for Evaluating Reasoning Traces","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Congchao Wang, Diwakar Singh, Mahdi Namazifar, Qiaozi Gao, Spyros Matsoukas, Yang Liu","submitted_at":"2026-08-12T21:00:22Z","abstract_excerpt":"Improving reasoning LLMs requires the ability to judge the quality of long reasoning traces for effective reasoning data curation, strong training signals during reinforcement learning, and an in-depth understanding of reasoning behaviors during model performance evaluation. Additionally, surfacing reasoning mistakes that the model makes would enable improving the model's performance at runtime through providing feedback. Due to the difficulty of this complex task on long reasoning traces, single-model judges (even frontier models) do not do well at identifying reasoning defects. Additionally,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.12585","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.12585/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-08-14T00:45:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gTsmZczL5pQ0m3BGuU1koX5zfzvbr/1Idtzvyf/jyKd5bfHIyY56ZPn2Xac/7tzEHOrf3DwsQr7RaRa2Dpi/DQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T08:57:17.050674Z"},"content_sha256":"ddd80e9cf5f52e7c95f7ec1fbd516ccf93d738a4714455028d026fea261f551a","schema_version":"1.0","event_id":"sha256:ddd80e9cf5f52e7c95f7ec1fbd516ccf93d738a4714455028d026fea261f551a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XBPOS36Y4FV3OH5FAKVBIUS5I4/bundle.json","state_url":"https://pith.science/pith/XBPOS36Y4FV3OH5FAKVBIUS5I4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XBPOS36Y4FV3OH5FAKVBIUS5I4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T08:57:17Z","links":{"resolver":"https://pith.science/pith/XBPOS36Y4FV3OH5FAKVBIUS5I4","bundle":"https://pith.science/pith/XBPOS36Y4FV3OH5FAKVBIUS5I4/bundle.json","state":"https://pith.science/pith/XBPOS36Y4FV3OH5FAKVBIUS5I4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XBPOS36Y4FV3OH5FAKVBIUS5I4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:XBPOS36Y4FV3OH5FAKVBIUS5I4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"da8251d17c898d8f4e6b1e9984f93f83792aab5c3953d0071fc52774b95da056","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-12T21:00:22Z","title_canon_sha256":"8f73098ec08a2aca09d12ed06323a04b128ffb450bbd2d9b66ad27c4752a0148"},"schema_version":"1.0","source":{"id":"2608.12585","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2608.12585","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"arxiv_version","alias_value":"2608.12585v1","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.12585","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"pith_short_12","alias_value":"XBPOS36Y4FV3","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"pith_short_16","alias_value":"XBPOS36Y4FV3OH5F","created_at":"2026-08-14T00:45:27Z"},{"alias_kind":"pith_short_8","alias_value":"XBPOS36Y","created_at":"2026-08-14T00:45:27Z"}],"graph_snapshots":[{"event_id":"sha256:ddd80e9cf5f52e7c95f7ec1fbd516ccf93d738a4714455028d026fea261f551a","target":"graph","created_at":"2026-08-14T00:45:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2608.12585/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Improving reasoning LLMs requires the ability to judge the quality of long reasoning traces for effective reasoning data curation, strong training signals during reinforcement learning, and an in-depth understanding of reasoning behaviors during model performance evaluation. Additionally, surfacing reasoning mistakes that the model makes would enable improving the model's performance at runtime through providing feedback. Due to the difficulty of this complex task on long reasoning traces, single-model judges (even frontier models) do not do well at identifying reasoning defects. Additionally,","authors_text":"Congchao Wang, Diwakar Singh, Mahdi Namazifar, Qiaozi Gao, Spyros Matsoukas, Yang Liu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-12T21:00:22Z","title":"Reasoning Jury: Multi-Model Consensus for Evaluating Reasoning Traces"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.12585","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:046667464531427a3fa5dbda35745a544997c079f4cf34a84d49727cff65776c","target":"record","created_at":"2026-08-14T00:45:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"da8251d17c898d8f4e6b1e9984f93f83792aab5c3953d0071fc52774b95da056","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-12T21:00:22Z","title_canon_sha256":"8f73098ec08a2aca09d12ed06323a04b128ffb450bbd2d9b66ad27c4752a0148"},"schema_version":"1.0","source":{"id":"2608.12585","kind":"arxiv","version":1}},"canonical_sha256":"b85ee96fd8e16bb71fa502aa14525d4712cc5ef37129c6ba8b5cb4c6783bdb35","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b85ee96fd8e16bb71fa502aa14525d4712cc5ef37129c6ba8b5cb4c6783bdb35","first_computed_at":"2026-08-14T00:45:27.941117Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-14T00:45:27.941117Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jMl57bGlime3HdlHbrX5U8hTwG/3UkQmr+3q9ER+dWO5Sw9DiKnwZvpUPhAWcBF9UT9sufkbWSWJWwRckVQZDg==","signature_status":"signed_v1","signed_at":"2026-08-14T00:45:27.951316Z","signed_message":"canonical_sha256_bytes"},"source_id":"2608.12585","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:046667464531427a3fa5dbda35745a544997c079f4cf34a84d49727cff65776c","sha256:ddd80e9cf5f52e7c95f7ec1fbd516ccf93d738a4714455028d026fea261f551a"],"state_sha256":"4ac53bf51e2f3bdb455ed9184ab955e719aba3f461bb428fb40b01821e96db30"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4Tl6kdCNuRWmzlJV4xoxGGnLKORn0VLecsau/GIVLVkufQpNC6G3GiQe44hbc2G7FnTfMFOgBg6APr51lH+4Aw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T08:57:17.055129Z","bundle_sha256":"1e62692c43da7fc4a5ed1fe818d8f8fd2e040142be66ec03ea9aaf92aafc0f01"}}