{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:YIPXLBFKIR64FLKSD77A5MQPXS","short_pith_number":"pith:YIPXLBFK","canonical_record":{"source":{"id":"2605.02395","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T09:36:57Z","cross_cats_sorted":[],"title_canon_sha256":"c149d2cbea3f31883b2110a96f47abfdbf165eee56ece73506d976a1e1b3bc08","abstract_canon_sha256":"0f4f310104438fccb75c5a90592ba01100485f452c56fb76290f326fd8fed9c9"},"schema_version":"1.0"},"canonical_sha256":"c21f7584aa447dc2ad521ffe0eb20fbcb742473a92848b82c03b10aea0fb069d","source":{"kind":"arxiv","id":"2605.02395","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.02395","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"arxiv_version","alias_value":"2605.02395v2","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.02395","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"pith_short_12","alias_value":"YIPXLBFKIR64","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"pith_short_16","alias_value":"YIPXLBFKIR64FLKS","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"pith_short_8","alias_value":"YIPXLBFK","created_at":"2026-06-05T01:14:39Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:YIPXLBFKIR64FLKSD77A5MQPXS","target":"record","payload":{"canonical_record":{"source":{"id":"2605.02395","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T09:36:57Z","cross_cats_sorted":[],"title_canon_sha256":"c149d2cbea3f31883b2110a96f47abfdbf165eee56ece73506d976a1e1b3bc08","abstract_canon_sha256":"0f4f310104438fccb75c5a90592ba01100485f452c56fb76290f326fd8fed9c9"},"schema_version":"1.0"},"canonical_sha256":"c21f7584aa447dc2ad521ffe0eb20fbcb742473a92848b82c03b10aea0fb069d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-05T01:14:39.903822Z","signature_b64":"uhK55P2UYzCl+GiCRPR21dAFwvF51Oio5sAeqYGXBus2soHC7yJc+eYaHHh3B5xT+ySJXf58tdo5fXCKp7+CDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c21f7584aa447dc2ad521ffe0eb20fbcb742473a92848b82c03b10aea0fb069d","last_reissued_at":"2026-06-05T01:14:39.903081Z","signature_status":"signed_v1","first_computed_at":"2026-06-05T01:14:39.903081Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.02395","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-05T01:14:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"EDrioRwKlFMRWGefzDiJO7lAUfK4IsJH33nqqJCVd41FTzMZ54mRgOLVrdIAiJqmcaQ1q3O6in84dFec5ZemBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T22:37:17.717899Z"},"content_sha256":"3d64872209a0dbfd2b9585de8f14440cbe1e5bf8ca573bb0928d96023d0a3f70","schema_version":"1.0","event_id":"sha256:3d64872209a0dbfd2b9585de8f14440cbe1e5bf8ca573bb0928d96023d0a3f70"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:YIPXLBFKIR64FLKSD77A5MQPXS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Controllable and Verifiable Process Data Synthesis for Process Reward Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A synthesis method builds controllable process supervision data by injecting template-aware errors into symbolic reasoning chains, recomputing trajectories, and translating them to natural language for training process reward models.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Lucien Wang, Yinghui Chi","submitted_at":"2026-05-04T09:36:57Z","abstract_excerpt":"Process reward models (PRMs) rely on high-quality process supervision data, yet existing construction methods often provide limited control over error location, error type, and trajectory consistency. We propose a controllable and verifiable framework for synthesizing process supervision data for PRMs. Our framework first constructs a correct symbolic reasoning chain, injects a template-aware error into an intermediate step, recomputes subsequent steps under the corrupted state, and verifies that the injected step is not derivable from its prefix. The resulting paired trajectories are prefix-i"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Experiments show that the synthesized data improve Best-of-8 reranking on logical reasoning benchmarks and transfer to mathematical reasoning. Step-level evaluation further shows that first-error localization remains substantially more challenging than overall step classification.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that template-aware errors injected into symbolic chains and then recomputed produce trajectories whose error patterns and consistency properties transfer meaningfully to natural-language reasoning processes used in real PRM training.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"A controllable synthesis method creates prefix-invalid yet trajectory-consistent process supervision data for training and evaluating process reward models by injecting verifiable errors into symbolic reasoning chains.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A synthesis method builds controllable process supervision data by injecting template-aware errors into symbolic reasoning chains, recomputing trajectories, and translating them to natural language for training process reward models.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"e97e014b772e7d5584e46bed9221eb708268a81dbbfc2074aee0a4ff959e3a09"},"source":{"id":"2605.02395","kind":"arxiv","version":2},"verdict":{"id":"fa652d8d-b563-4371-8985-7f40c490bd10","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-08T19:17:57.475971Z","strongest_claim":"Experiments show that the synthesized data improve Best-of-8 reranking on logical reasoning benchmarks and transfer to mathematical reasoning. Step-level evaluation further shows that first-error localization remains substantially more challenging than overall step classification.","one_line_summary":"A controllable synthesis method creates prefix-invalid yet trajectory-consistent process supervision data for training and evaluating process reward models by injecting verifiable errors into symbolic reasoning chains.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that template-aware errors injected into symbolic chains and then recomputed produce trajectories whose error patterns and consistency properties transfer meaningfully to natural-language reasoning processes used in real PRM training.","pith_extraction_headline":"A synthesis method builds controllable process supervision data by injecting template-aware errors into symbolic reasoning chains, recomputing trajectories, and translating them to natural language for training process reward models."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.02395/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-20T15:41:17.794272Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_title_agreement","ran_at":"2026-05-20T03:31:22.343178Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T16:23:50.552151Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"e756a40954c2b1d2ba5d843675f18a1435c63789fce239752af9dd339f6d9b4d"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":1,"snapshot_sha256":"292dacf6e591a6feb3c7dec02a20c8212649c4b11cbed6ffc6ddab7609832898"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"fa652d8d-b563-4371-8985-7f40c490bd10"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-05T01:14:39Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WhF8Razsb+qotm9AbBsFXhUiDsaClBHFjQfAK6thlxBFRfNIlKLq4VIfXkunDLrKU5uTBigHODiB7EWFTJlZBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T22:37:17.718572Z"},"content_sha256":"cc0b97cf43f95b60154a25c0ceb2b410f268db70fa426c2a85f303dcbb646077","schema_version":"1.0","event_id":"sha256:cc0b97cf43f95b60154a25c0ceb2b410f268db70fa426c2a85f303dcbb646077"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YIPXLBFKIR64FLKSD77A5MQPXS/bundle.json","state_url":"https://pith.science/pith/YIPXLBFKIR64FLKSD77A5MQPXS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YIPXLBFKIR64FLKSD77A5MQPXS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T22:37:17Z","links":{"resolver":"https://pith.science/pith/YIPXLBFKIR64FLKSD77A5MQPXS","bundle":"https://pith.science/pith/YIPXLBFKIR64FLKSD77A5MQPXS/bundle.json","state":"https://pith.science/pith/YIPXLBFKIR64FLKSD77A5MQPXS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YIPXLBFKIR64FLKSD77A5MQPXS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:YIPXLBFKIR64FLKSD77A5MQPXS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0f4f310104438fccb75c5a90592ba01100485f452c56fb76290f326fd8fed9c9","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T09:36:57Z","title_canon_sha256":"c149d2cbea3f31883b2110a96f47abfdbf165eee56ece73506d976a1e1b3bc08"},"schema_version":"1.0","source":{"id":"2605.02395","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.02395","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"arxiv_version","alias_value":"2605.02395v2","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.02395","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"pith_short_12","alias_value":"YIPXLBFKIR64","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"pith_short_16","alias_value":"YIPXLBFKIR64FLKS","created_at":"2026-06-05T01:14:39Z"},{"alias_kind":"pith_short_8","alias_value":"YIPXLBFK","created_at":"2026-06-05T01:14:39Z"}],"graph_snapshots":[{"event_id":"sha256:cc0b97cf43f95b60154a25c0ceb2b410f268db70fa426c2a85f303dcbb646077","target":"graph","created_at":"2026-06-05T01:14:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Experiments show that the synthesized data improve Best-of-8 reranking on logical reasoning benchmarks and transfer to mathematical reasoning. Step-level evaluation further shows that first-error localization remains substantially more challenging than overall step classification."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The assumption that template-aware errors injected into symbolic chains and then recomputed produce trajectories whose error patterns and consistency properties transfer meaningfully to natural-language reasoning processes used in real PRM training."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"A controllable synthesis method creates prefix-invalid yet trajectory-consistent process supervision data for training and evaluating process reward models by injecting verifiable errors into symbolic reasoning chains."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"A synthesis method builds controllable process supervision data by injecting template-aware errors into symbolic reasoning chains, recomputing trajectories, and translating them to natural language for training process reward models."}],"snapshot_sha256":"e97e014b772e7d5584e46bed9221eb708268a81dbbfc2074aee0a4ff959e3a09"},"formal_canon":{"evidence_count":1,"snapshot_sha256":"292dacf6e591a6feb3c7dec02a20c8212649c4b11cbed6ffc6ddab7609832898"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T15:41:17.794272Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_title_agreement","ran_at":"2026-05-20T03:31:22.343178Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T16:23:50.552151Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2605.02395/integrity.json","findings":[],"snapshot_sha256":"e756a40954c2b1d2ba5d843675f18a1435c63789fce239752af9dd339f6d9b4d","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Process reward models (PRMs) rely on high-quality process supervision data, yet existing construction methods often provide limited control over error location, error type, and trajectory consistency. We propose a controllable and verifiable framework for synthesizing process supervision data for PRMs. Our framework first constructs a correct symbolic reasoning chain, injects a template-aware error into an intermediate step, recomputes subsequent steps under the corrupted state, and verifies that the injected step is not derivable from its prefix. The resulting paired trajectories are prefix-i","authors_text":"Lucien Wang, Yinghui Chi","cross_cats":[],"headline":"A synthesis method builds controllable process supervision data by injecting template-aware errors into symbolic reasoning chains, recomputing trajectories, and translating them to natural language for training process reward models.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T09:36:57Z","title":"Controllable and Verifiable Process Data Synthesis for Process Reward Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.02395","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-08T19:17:57.475971Z","id":"fa652d8d-b563-4371-8985-7f40c490bd10","model_set":{"reader":"grok-4.3"},"one_line_summary":"A controllable synthesis method creates prefix-invalid yet trajectory-consistent process supervision data for training and evaluating process reward models by injecting verifiable errors into symbolic reasoning chains.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"A synthesis method builds controllable process supervision data by injecting template-aware errors into symbolic reasoning chains, recomputing trajectories, and translating them to natural language for training process reward models.","strongest_claim":"Experiments show that the synthesized data improve Best-of-8 reranking on logical reasoning benchmarks and transfer to mathematical reasoning. Step-level evaluation further shows that first-error localization remains substantially more challenging than overall step classification.","weakest_assumption":"The assumption that template-aware errors injected into symbolic chains and then recomputed produce trajectories whose error patterns and consistency properties transfer meaningfully to natural-language reasoning processes used in real PRM training."}},"verdict_id":"fa652d8d-b563-4371-8985-7f40c490bd10"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3d64872209a0dbfd2b9585de8f14440cbe1e5bf8ca573bb0928d96023d0a3f70","target":"record","created_at":"2026-06-05T01:14:39Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0f4f310104438fccb75c5a90592ba01100485f452c56fb76290f326fd8fed9c9","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T09:36:57Z","title_canon_sha256":"c149d2cbea3f31883b2110a96f47abfdbf165eee56ece73506d976a1e1b3bc08"},"schema_version":"1.0","source":{"id":"2605.02395","kind":"arxiv","version":2}},"canonical_sha256":"c21f7584aa447dc2ad521ffe0eb20fbcb742473a92848b82c03b10aea0fb069d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c21f7584aa447dc2ad521ffe0eb20fbcb742473a92848b82c03b10aea0fb069d","first_computed_at":"2026-06-05T01:14:39.903081Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-05T01:14:39.903081Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"uhK55P2UYzCl+GiCRPR21dAFwvF51Oio5sAeqYGXBus2soHC7yJc+eYaHHh3B5xT+ySJXf58tdo5fXCKp7+CDg==","signature_status":"signed_v1","signed_at":"2026-06-05T01:14:39.903822Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.02395","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3d64872209a0dbfd2b9585de8f14440cbe1e5bf8ca573bb0928d96023d0a3f70","sha256:cc0b97cf43f95b60154a25c0ceb2b410f268db70fa426c2a85f303dcbb646077"],"state_sha256":"8ff67e4e633bc90d5f0e4b9c9e7f339237337384649fc3edfd3bf8170082b7dc"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qMee5n9BsK/2PwG9QuDxlwyigj8E3GNP0gwldnFJ49ODJjMZX4h8LuYhsE6KUd4juIKhoTvbF1vRplZCU6OxCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T22:37:17.723114Z","bundle_sha256":"7ae3e9aa10d78eb2582f05c1a8d188fc012aee5c1206146d5926dcb4f0c4f99b"}}