{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EJL353UFBAVJ5ANVTZB2PHPFFI","short_pith_number":"pith:EJL353UF","schema_version":"1.0","canonical_sha256":"2257beee85082a9e81b59e43a79de52a204bc2387c46301487b3685f13af742d","source":{"kind":"arxiv","id":"2505.07889","version":4},"attestation_state":"computed","paper":{"title":"BioProBench: A Corpus and Benchmark for Biological Protocol Reasoning in Autonomous Science","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jingya Wang Li Yuan, Liuzhenghao Lv, Xiancheng Zhang, Yonghong Tian, Yuyang Liu","submitted_at":"2025-05-11T09:42:24Z","abstract_excerpt":"The realization of autonomous scientific experimentation is currently limited by LLMs' struggle to grasp the strict procedural logic and accuracy required by biological protocols. To address this fundamental challenge, we present \\textbf{BioProBench}, a comprehensive resource for procedural reasoning in biology. BioProBench is grounded in \\textbf{BioProCorpus}, a foundational collection of 22,413 human-written protocols. From this corpus, we systematically constructed a dataset of 523,784 task instances, offering both a large-scale training resource and a rigorous benchmark with novel metrics."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.07889","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-11T09:42:24Z","cross_cats_sorted":[],"title_canon_sha256":"5364144b2b7a6456c149ee4a45150401a027d108bd3df7ebec9f6450fd48070b","abstract_canon_sha256":"80ad62fc0dc1e9aa4eafa961025d1f5724b13b5183192b9dddf9bfe3df817bd9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T01:22:19.981115Z","signature_b64":"ldGLpHriYAe4ZEk8MC0cMeh+yzgpRgJCA5JZSniVUWHvhOxad5Yd+Nqb01xzs5NLb7T1lNNpQDAkyF7qnckPCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2257beee85082a9e81b59e43a79de52a204bc2387c46301487b3685f13af742d","last_reissued_at":"2026-07-28T01:22:19.980058Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T01:22:19.980058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BioProBench: A Corpus and Benchmark for Biological Protocol Reasoning in Autonomous Science","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jingya Wang Li Yuan, Liuzhenghao Lv, Xiancheng Zhang, Yonghong Tian, Yuyang Liu","submitted_at":"2025-05-11T09:42:24Z","abstract_excerpt":"The realization of autonomous scientific experimentation is currently limited by LLMs' struggle to grasp the strict procedural logic and accuracy required by biological protocols. To address this fundamental challenge, we present \\textbf{BioProBench}, a comprehensive resource for procedural reasoning in biology. BioProBench is grounded in \\textbf{BioProCorpus}, a foundational collection of 22,413 human-written protocols. From this corpus, we systematically constructed a dataset of 523,784 task instances, offering both a large-scale training resource and a rigorous benchmark with novel metrics."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.07889","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.07889/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.07889","created_at":"2026-07-28T01:22:19.980538+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.07889v4","created_at":"2026-07-28T01:22:19.980538+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.07889","created_at":"2026-07-28T01:22:19.980538+00:00"},{"alias_kind":"pith_short_12","alias_value":"EJL353UFBAVJ","created_at":"2026-07-28T01:22:19.980538+00:00"},{"alias_kind":"pith_short_16","alias_value":"EJL353UFBAVJ5ANV","created_at":"2026-07-28T01:22:19.980538+00:00"},{"alias_kind":"pith_short_8","alias_value":"EJL353UF","created_at":"2026-07-28T01:22:19.980538+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":5,"sample":[{"citing_arxiv_id":"2606.08035","citing_title":"DyCo-RL: Dynamic Cross-Modal Coordination for Visual Reasoning","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04579","citing_title":"SCI-PRM: A Tool Aware Process Reward Model for Scientific Reasoning Verification","ref_index":54,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01145","citing_title":"Reasoning4Sciences: Bridging Reasoning Language Models to All Scientific Branches","ref_index":169,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01145","citing_title":"Reasoning4Sciences: Bridging Reasoning Language Models to All Scientific Branches","ref_index":182,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15766","citing_title":"BioXArena: Benchmarking LLM Agents on Multi-Modal Biomedical Machine Learning Tasks","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI","json":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI.json","graph_json":"https://pith.science/api/pith-number/EJL353UFBAVJ5ANVTZB2PHPFFI/graph.json","events_json":"https://pith.science/api/pith-number/EJL353UFBAVJ5ANVTZB2PHPFFI/events.json","paper":"https://pith.science/paper/EJL353UF"},"agent_actions":{"view_html":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI","download_json":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI.json","view_paper":"https://pith.science/paper/EJL353UF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.07889&json=true","fetch_graph":"https://pith.science/api/pith-number/EJL353UFBAVJ5ANVTZB2PHPFFI/graph.json","fetch_events":"https://pith.science/api/pith-number/EJL353UFBAVJ5ANVTZB2PHPFFI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI/action/storage_attestation","attest_author":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI/action/author_attestation","sign_citation":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI/action/citation_signature","submit_replication":"https://pith.science/pith/EJL353UFBAVJ5ANVTZB2PHPFFI/action/replication_record"}},"created_at":"2026-07-28T01:22:19.980538+00:00","updated_at":"2026-07-28T01:22:19.980538+00:00"}