{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:6PMTCRO53VPEZBR2U4JX4JREVR","short_pith_number":"pith:6PMTCRO5","schema_version":"1.0","canonical_sha256":"f3d93145dddd5e4c863aa7137e2624ac4e8060798d62ae30760a1a5b5e71df6a","source":{"kind":"arxiv","id":"2608.07437","version":1},"attestation_state":"computed","paper":{"title":"Fisher-R1: Training LLM Agents for Reliable Hypothesis Testing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Guanhua Chen, James Zou, Jiacheng Miao, Jin Mu","submitted_at":"2026-08-07T17:22:00Z","abstract_excerpt":"Reliable hypothesis testing is the foundation of many empirical scientific claims. Large language model (LLM) agents are increasingly used to automate this process, as they can inspect datasets, generate code, and produce analyses end-to-end. However, we show that they frequently make subtle inferential errors that lead to incorrect conclusions despite correctly executed analyses. Existing benchmarks fail to capture this failure mode, as they rarely assess whether a reported p-value is statistically valid given the assumptions underlying the data. We address this gap by building P-Bench, a ben"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.07437","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-08-07T17:22:00Z","cross_cats_sorted":[],"title_canon_sha256":"be6c2db6933561c1addda8dbc14fe3fb328d0147db0bf900b5176e6444f58d0b","abstract_canon_sha256":"6b3d626d3abd5b0773cd076e1891de7e35d156e1ca8410dbdb9718d432e31fc7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-10T01:15:23.644290Z","signature_b64":"CUCoGls1X3PQlqJ8khRLcw9WOeDNbfMES6g6Eweb5nXS8PWn5ZrVPm/GdxOghnF+jHDcRo5ZfUVfM5YjDhdECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3d93145dddd5e4c863aa7137e2624ac4e8060798d62ae30760a1a5b5e71df6a","last_reissued_at":"2026-08-10T01:15:23.641810Z","signature_status":"signed_v1","first_computed_at":"2026-08-10T01:15:23.641810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fisher-R1: Training LLM Agents for Reliable Hypothesis Testing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Guanhua Chen, James Zou, Jiacheng Miao, Jin Mu","submitted_at":"2026-08-07T17:22:00Z","abstract_excerpt":"Reliable hypothesis testing is the foundation of many empirical scientific claims. Large language model (LLM) agents are increasingly used to automate this process, as they can inspect datasets, generate code, and produce analyses end-to-end. However, we show that they frequently make subtle inferential errors that lead to incorrect conclusions despite correctly executed analyses. Existing benchmarks fail to capture this failure mode, as they rarely assess whether a reported p-value is statistically valid given the assumptions underlying the data. We address this gap by building P-Bench, a ben"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.07437","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.07437/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.07437","created_at":"2026-08-10T01:15:23.642951+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.07437v1","created_at":"2026-08-10T01:15:23.642951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.07437","created_at":"2026-08-10T01:15:23.642951+00:00"},{"alias_kind":"pith_short_12","alias_value":"6PMTCRO53VPE","created_at":"2026-08-10T01:15:23.642951+00:00"},{"alias_kind":"pith_short_16","alias_value":"6PMTCRO53VPEZBR2","created_at":"2026-08-10T01:15:23.642951+00:00"},{"alias_kind":"pith_short_8","alias_value":"6PMTCRO5","created_at":"2026-08-10T01:15:23.642951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR","json":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR.json","graph_json":"https://pith.science/api/pith-number/6PMTCRO53VPEZBR2U4JX4JREVR/graph.json","events_json":"https://pith.science/api/pith-number/6PMTCRO53VPEZBR2U4JX4JREVR/events.json","paper":"https://pith.science/paper/6PMTCRO5"},"agent_actions":{"view_html":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR","download_json":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR.json","view_paper":"https://pith.science/paper/6PMTCRO5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.07437&json=true","fetch_graph":"https://pith.science/api/pith-number/6PMTCRO53VPEZBR2U4JX4JREVR/graph.json","fetch_events":"https://pith.science/api/pith-number/6PMTCRO53VPEZBR2U4JX4JREVR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR/action/storage_attestation","attest_author":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR/action/author_attestation","sign_citation":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR/action/citation_signature","submit_replication":"https://pith.science/pith/6PMTCRO53VPEZBR2U4JX4JREVR/action/replication_record"}},"created_at":"2026-08-10T01:15:23.642951+00:00","updated_at":"2026-08-10T01:15:23.642951+00:00"}