{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FIM5QBGWX3Y6JITJQUO34LXIMD","short_pith_number":"pith:FIM5QBGW","schema_version":"1.0","canonical_sha256":"2a19d804d6bef1e4a269851dbe2ee860d63c0e7485c799f30bac2a11a14c1f96","source":{"kind":"arxiv","id":"2507.04766","version":1},"attestation_state":"computed","paper":{"title":"ABench-Physics: Benchmarking Physical Reasoning in LLMs via High-Difficulty and Dynamic Physics Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bowen Song, Chao Huang, Cheng Lin, Feng Wang, Junbo Zhao, Yanmei Gu, Yihong Zhuang, Yiming Zhang, Yingfan Ma, Yuanyuan Wang, Zenan Huang, Zhengkai Yang","submitted_at":"2025-07-07T08:43:56Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive performance in domains such as mathematics and programming, yet their capabilities in physics remain underexplored and poorly understood. Physics poses unique challenges that demand not only precise computation but also deep conceptual understanding and physical modeling skills. Existing benchmarks often fall short due to limited difficulty, multiple-choice formats, and static evaluation settings that fail to capture physical modeling ability. In this paper, we introduce ABench-Physics, a novel benchmark designed to rigorously evaluate LLMs' p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.04766","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-07T08:43:56Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"6aadc6ba7cb54534b442d1bbcbd34f212a28f7db8ec10c72b52baf9112498469","abstract_canon_sha256":"4665e978f50243af098b0ded256ecd969a97caeda5cc6347d905443760352822"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:33:26.802197Z","signature_b64":"INAoPeSzhmEnia4fJg5YaqEsR+iWX0SVLJ3iMGm/0TBgc+pFCf4NHHpEJodOfbSl0adNPmGDBnGA/qtWydS+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a19d804d6bef1e4a269851dbe2ee860d63c0e7485c799f30bac2a11a14c1f96","last_reissued_at":"2026-07-05T11:33:26.801690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:33:26.801690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ABench-Physics: Benchmarking Physical Reasoning in LLMs via High-Difficulty and Dynamic Physics Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bowen Song, Chao Huang, Cheng Lin, Feng Wang, Junbo Zhao, Yanmei Gu, Yihong Zhuang, Yiming Zhang, Yingfan Ma, Yuanyuan Wang, Zenan Huang, Zhengkai Yang","submitted_at":"2025-07-07T08:43:56Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive performance in domains such as mathematics and programming, yet their capabilities in physics remain underexplored and poorly understood. Physics poses unique challenges that demand not only precise computation but also deep conceptual understanding and physical modeling skills. Existing benchmarks often fall short due to limited difficulty, multiple-choice formats, and static evaluation settings that fail to capture physical modeling ability. In this paper, we introduce ABench-Physics, a novel benchmark designed to rigorously evaluate LLMs' p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.04766","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.04766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.04766","created_at":"2026-07-05T11:33:26.801764+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.04766v1","created_at":"2026-07-05T11:33:26.801764+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.04766","created_at":"2026-07-05T11:33:26.801764+00:00"},{"alias_kind":"pith_short_12","alias_value":"FIM5QBGWX3Y6","created_at":"2026-07-05T11:33:26.801764+00:00"},{"alias_kind":"pith_short_16","alias_value":"FIM5QBGWX3Y6JITJ","created_at":"2026-07-05T11:33:26.801764+00:00"},{"alias_kind":"pith_short_8","alias_value":"FIM5QBGW","created_at":"2026-07-05T11:33:26.801764+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00276","citing_title":"Testing Frontier Large Language Models' Physics Literacy in Parallel Physical Worlds","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12549","citing_title":"The Serial Scaling Hypothesis","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07064","citing_title":"OmniFysics: Towards Physical Intelligence Evolution via Omni-Modal Signal Processing and Network Optimization","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18936","citing_title":"Fine-Tuning Small Reasoning Models for Quantum Field Theory","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17928","citing_title":"HEALing Entropy Collapse: Enhancing Exploration in Few-Shot RLVR via Hybrid-Domain Entropy Dynamics Alignment","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD","json":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD.json","graph_json":"https://pith.science/api/pith-number/FIM5QBGWX3Y6JITJQUO34LXIMD/graph.json","events_json":"https://pith.science/api/pith-number/FIM5QBGWX3Y6JITJQUO34LXIMD/events.json","paper":"https://pith.science/paper/FIM5QBGW"},"agent_actions":{"view_html":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD","download_json":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD.json","view_paper":"https://pith.science/paper/FIM5QBGW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.04766&json=true","fetch_graph":"https://pith.science/api/pith-number/FIM5QBGWX3Y6JITJQUO34LXIMD/graph.json","fetch_events":"https://pith.science/api/pith-number/FIM5QBGWX3Y6JITJQUO34LXIMD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD/action/storage_attestation","attest_author":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD/action/author_attestation","sign_citation":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD/action/citation_signature","submit_replication":"https://pith.science/pith/FIM5QBGWX3Y6JITJQUO34LXIMD/action/replication_record"}},"created_at":"2026-07-05T11:33:26.801764+00:00","updated_at":"2026-07-05T11:33:26.801764+00:00"}