{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:NJVNSFNVVR5OYWYFXIQC772ESY","short_pith_number":"pith:NJVNSFNV","canonical_record":{"source":{"id":"2604.24198","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-04-27T09:00:30Z","cross_cats_sorted":["cs.AI","cs.CE","cs.LG","cs.MA"],"title_canon_sha256":"45172b74396e67fd503eb7c2e5237f548e9d36c5d218da62f41e187452af1002","abstract_canon_sha256":"ce10f76542bdf90fa13fd1e77811802f36b8ad814a1003589a35a368a42c41ae"},"schema_version":"1.0"},"canonical_sha256":"6a6ad915b5ac7aec5b05ba202fff4496298529e81f85baff66f9f27cace4cfb6","source":{"kind":"arxiv","id":"2604.24198","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.24198","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"arxiv_version","alias_value":"2604.24198v2","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.24198","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"pith_short_12","alias_value":"NJVNSFNVVR5O","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"pith_short_16","alias_value":"NJVNSFNVVR5OYWYF","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"pith_short_8","alias_value":"NJVNSFNV","created_at":"2026-06-23T02:12:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:NJVNSFNVVR5OYWYFXIQC772ESY","target":"record","payload":{"canonical_record":{"source":{"id":"2604.24198","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-04-27T09:00:30Z","cross_cats_sorted":["cs.AI","cs.CE","cs.LG","cs.MA"],"title_canon_sha256":"45172b74396e67fd503eb7c2e5237f548e9d36c5d218da62f41e187452af1002","abstract_canon_sha256":"ce10f76542bdf90fa13fd1e77811802f36b8ad814a1003589a35a368a42c41ae"},"schema_version":"1.0"},"canonical_sha256":"6a6ad915b5ac7aec5b05ba202fff4496298529e81f85baff66f9f27cace4cfb6","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T02:12:49.415042Z","signature_b64":"219LD258NGKw67WeNPp0+opUXV7b7JppajubsalvoIbMBTOC4XOKuPx4uluomClow4KPa9Y/7bz0fY4OeOSKAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a6ad915b5ac7aec5b05ba202fff4496298529e81f85baff66f9f27cace4cfb6","last_reissued_at":"2026-06-23T02:12:49.414552Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T02:12:49.414552Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.24198","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-23T02:12:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jI1a9aIjfZkBcNz9QLoeOYsYdMjF7dyFIe7ZwhSgsbI6lxj/LX+8G5tWDEHpuwJO0w0Nq0GvZeaNwBcw+8lzCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T05:22:16.401638Z"},"content_sha256":"5bf90610bde2619a3e04b30c7608c6245d1a9d013b37741627c4cc79a55daa44","schema_version":"1.0","event_id":"sha256:5bf90610bde2619a3e04b30c7608c6245d1a9d013b37741627c4cc79a55daa44"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:NJVNSFNVVR5OYWYFXIQC772ESY","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Rewarding the Scientific Process: Process-Level Reward Modeling for Agentic Data Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"DataPRM improves AI data analysis agents by actively probing execution states to catch silent errors and applying ternary rewards that separate fixable mistakes from fatal ones.","cross_cats":["cs.AI","cs.CE","cs.LG","cs.MA"],"primary_cat":"cs.CL","authors_text":"Huajun Chen, Kewei Xu, Lun Du, Ningyu Zhang, Shuofei Qiao, Yuqi Zhu, Zhisong Qiu","submitted_at":"2026-04-27T09:00:30Z","abstract_excerpt":"Process Reward Models (PRMs) have achieved remarkable success in augmenting the reasoning capabilities of Large Language Models (LLMs) within static domains such as mathematics. However, their potential in dynamic data analysis tasks remains underexplored. In this work, we first present a empirical study revealing that general-domain PRMs struggle to supervise data analysis agents. Specifically, they fail to detect silent errors, logical flaws that yield incorrect results without triggering interpreter exceptions, and erroneously penalize exploratory actions, mistaking necessary trial-and-erro"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"DataPRM improves downstream policy LLMs by 7.21% on ScienceAgentBench and 11.28% on DABStep using Best-of-N inference; integrating DataPRM into Reinforcement Learning yields 78.73% on DABench and 64.84% on TableBench.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the performance gains are attributable to the environment-aware probing and reflection-aware ternary strategy rather than to the quality or diversity of the 8K training instances or to unstated differences in baselines and evaluation protocols.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"DataPRM is a new process reward model for data analysis agents that detects silent errors via environment interaction and ternary rewards, yielding 7-11% gains on benchmarks and further RL improvements.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"DataPRM improves AI data analysis agents by actively probing execution states to catch silent errors and applying ternary rewards that separate fixable mistakes from fatal ones.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"a33d23501240dcdf950819dc1dfc690077e848b0ef53a6208ec4d7516803d0b0"},"source":{"id":"2604.24198","kind":"arxiv","version":2},"verdict":{"id":"9bd0cc45-c736-4f90-b6f3-539fe6318dff","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-08T03:43:41.171135Z","strongest_claim":"DataPRM improves downstream policy LLMs by 7.21% on ScienceAgentBench and 11.28% on DABStep using Best-of-N inference; integrating DataPRM into Reinforcement Learning yields 78.73% on DABench and 64.84% on TableBench.","one_line_summary":"DataPRM is a new process reward model for data analysis agents that detects silent errors via environment interaction and ternary rewards, yielding 7-11% gains on benchmarks and further RL improvements.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the performance gains are attributable to the environment-aware probing and reflection-aware ternary strategy rather than to the quality or diversity of the 8K training instances or to unstated differences in baselines and evaluation protocols.","pith_extraction_headline":"DataPRM improves AI data analysis agents by actively probing execution states to catch silent errors and applying ternary rewards that separate fixable mistakes from fatal ones."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.24198/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-21T07:36:17.589422Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T22:22:53.249847Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"fe11f4bda85d1d7275993023b51a9ea1b18a154349dffffe1c080d5149ec4148"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"9bd0cc45-c736-4f90-b6f3-539fe6318dff"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-23T02:12:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Sgv1WFkFULxYiAChCLO7SxNXIierWRpDU+e8DR6PgeqWzooPJBXk0ScFeRJzJ6vRf7JCbk4a+lDcKAQWbk3zBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T05:22:16.402117Z"},"content_sha256":"3f6bf7cf5ca42b02ae4c707ab6a4ee32e1a4564a0a21bd7cc44ce12f50220e94","schema_version":"1.0","event_id":"sha256:3f6bf7cf5ca42b02ae4c707ab6a4ee32e1a4564a0a21bd7cc44ce12f50220e94"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NJVNSFNVVR5OYWYFXIQC772ESY/bundle.json","state_url":"https://pith.science/pith/NJVNSFNVVR5OYWYFXIQC772ESY/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NJVNSFNVVR5OYWYFXIQC772ESY/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-22T05:22:16Z","links":{"resolver":"https://pith.science/pith/NJVNSFNVVR5OYWYFXIQC772ESY","bundle":"https://pith.science/pith/NJVNSFNVVR5OYWYFXIQC772ESY/bundle.json","state":"https://pith.science/pith/NJVNSFNVVR5OYWYFXIQC772ESY/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NJVNSFNVVR5OYWYFXIQC772ESY/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:NJVNSFNVVR5OYWYFXIQC772ESY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ce10f76542bdf90fa13fd1e77811802f36b8ad814a1003589a35a368a42c41ae","cross_cats_sorted":["cs.AI","cs.CE","cs.LG","cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-04-27T09:00:30Z","title_canon_sha256":"45172b74396e67fd503eb7c2e5237f548e9d36c5d218da62f41e187452af1002"},"schema_version":"1.0","source":{"id":"2604.24198","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.24198","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"arxiv_version","alias_value":"2604.24198v2","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.24198","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"pith_short_12","alias_value":"NJVNSFNVVR5O","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"pith_short_16","alias_value":"NJVNSFNVVR5OYWYF","created_at":"2026-06-23T02:12:49Z"},{"alias_kind":"pith_short_8","alias_value":"NJVNSFNV","created_at":"2026-06-23T02:12:49Z"}],"graph_snapshots":[{"event_id":"sha256:3f6bf7cf5ca42b02ae4c707ab6a4ee32e1a4564a0a21bd7cc44ce12f50220e94","target":"graph","created_at":"2026-06-23T02:12:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"DataPRM improves downstream policy LLMs by 7.21% on ScienceAgentBench and 11.28% on DABStep using Best-of-N inference; integrating DataPRM into Reinforcement Learning yields 78.73% on DABench and 64.84% on TableBench."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the performance gains are attributable to the environment-aware probing and reflection-aware ternary strategy rather than to the quality or diversity of the 8K training instances or to unstated differences in baselines and evaluation protocols."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"DataPRM is a new process reward model for data analysis agents that detects silent errors via environment interaction and ternary rewards, yielding 7-11% gains on benchmarks and further RL improvements."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"DataPRM improves AI data analysis agents by actively probing execution states to catch silent errors and applying ternary rewards that separate fixable mistakes from fatal ones."}],"snapshot_sha256":"a33d23501240dcdf950819dc1dfc690077e848b0ef53a6208ec4d7516803d0b0"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-21T07:36:17.589422Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T22:22:53.249847Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.24198/integrity.json","findings":[],"snapshot_sha256":"fe11f4bda85d1d7275993023b51a9ea1b18a154349dffffe1c080d5149ec4148","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Process Reward Models (PRMs) have achieved remarkable success in augmenting the reasoning capabilities of Large Language Models (LLMs) within static domains such as mathematics. However, their potential in dynamic data analysis tasks remains underexplored. In this work, we first present a empirical study revealing that general-domain PRMs struggle to supervise data analysis agents. Specifically, they fail to detect silent errors, logical flaws that yield incorrect results without triggering interpreter exceptions, and erroneously penalize exploratory actions, mistaking necessary trial-and-erro","authors_text":"Huajun Chen, Kewei Xu, Lun Du, Ningyu Zhang, Shuofei Qiao, Yuqi Zhu, Zhisong Qiu","cross_cats":["cs.AI","cs.CE","cs.LG","cs.MA"],"headline":"DataPRM improves AI data analysis agents by actively probing execution states to catch silent errors and applying ternary rewards that separate fixable mistakes from fatal ones.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-04-27T09:00:30Z","title":"Rewarding the Scientific Process: Process-Level Reward Modeling for Agentic Data Analysis"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.24198","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-08T03:43:41.171135Z","id":"9bd0cc45-c736-4f90-b6f3-539fe6318dff","model_set":{"reader":"grok-4.3"},"one_line_summary":"DataPRM is a new process reward model for data analysis agents that detects silent errors via environment interaction and ternary rewards, yielding 7-11% gains on benchmarks and further RL improvements.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"DataPRM improves AI data analysis agents by actively probing execution states to catch silent errors and applying ternary rewards that separate fixable mistakes from fatal ones.","strongest_claim":"DataPRM improves downstream policy LLMs by 7.21% on ScienceAgentBench and 11.28% on DABStep using Best-of-N inference; integrating DataPRM into Reinforcement Learning yields 78.73% on DABench and 64.84% on TableBench.","weakest_assumption":"That the performance gains are attributable to the environment-aware probing and reflection-aware ternary strategy rather than to the quality or diversity of the 8K training instances or to unstated differences in baselines and evaluation protocols."}},"verdict_id":"9bd0cc45-c736-4f90-b6f3-539fe6318dff"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5bf90610bde2619a3e04b30c7608c6245d1a9d013b37741627c4cc79a55daa44","target":"record","created_at":"2026-06-23T02:12:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ce10f76542bdf90fa13fd1e77811802f36b8ad814a1003589a35a368a42c41ae","cross_cats_sorted":["cs.AI","cs.CE","cs.LG","cs.MA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-04-27T09:00:30Z","title_canon_sha256":"45172b74396e67fd503eb7c2e5237f548e9d36c5d218da62f41e187452af1002"},"schema_version":"1.0","source":{"id":"2604.24198","kind":"arxiv","version":2}},"canonical_sha256":"6a6ad915b5ac7aec5b05ba202fff4496298529e81f85baff66f9f27cace4cfb6","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6a6ad915b5ac7aec5b05ba202fff4496298529e81f85baff66f9f27cace4cfb6","first_computed_at":"2026-06-23T02:12:49.414552Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-23T02:12:49.414552Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"219LD258NGKw67WeNPp0+opUXV7b7JppajubsalvoIbMBTOC4XOKuPx4uluomClow4KPa9Y/7bz0fY4OeOSKAg==","signature_status":"signed_v1","signed_at":"2026-06-23T02:12:49.415042Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.24198","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5bf90610bde2619a3e04b30c7608c6245d1a9d013b37741627c4cc79a55daa44","sha256:3f6bf7cf5ca42b02ae4c707ab6a4ee32e1a4564a0a21bd7cc44ce12f50220e94"],"state_sha256":"8abdbceaefa13ad84021d740069e523015c507ef420afc31dd949044055485ac"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mOiSSgWhGvUS8a2iADj19ow3MGLBqT5WCKMAoasis9wXyjYNiQyu0xZ/mwSLagzzSV63wLyw+mDbbMg+vyAaDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-22T05:22:16.405001Z","bundle_sha256":"682379541f7f49d7d4320b714b93c32d1ce19c803714b6062b0cc334c187c368"}}