{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:E57R7TF6VBRILNSTTYFBU4CPFW","short_pith_number":"pith:E57R7TF6","schema_version":"1.0","canonical_sha256":"277f1fccbea86285b6539e0a1a704f2d9424a38ffa6c199dbe2ac28583aa3d77","source":{"kind":"arxiv","id":"2502.00334","version":4},"attestation_state":"computed","paper":{"title":"UGPhysics: A Comprehensive Benchmark for Undergraduate Physics Reasoning with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Can Yang, Jiaxin Zhang, Qiyun Xu, Shizhe Diao, Tianhao Chen, Tong Xiao, Xin Xu, Yang Wang, Yuchen Yan","submitted_at":"2025-02-01T06:42:02Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in solving complex reasoning tasks, particularly in mathematics. However, the domain of physics reasoning presents unique challenges that have received significantly less attention. Existing benchmarks often fall short in evaluating LLMs' abilities on the breadth and depth of undergraduate-level physics, underscoring the need for a comprehensive evaluation. To fill this gap, we introduce UGPhysics, a large-scale and comprehensive benchmark specifically designed to evaluate UnderGraduate-level Physics (UGPhysics) reasoning w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.00334","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-01T06:42:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4875d868d3661a3df972f7b3a98d5495e8e571e29665f64dcdd6572a1a944190","abstract_canon_sha256":"99e4cf276e8b3bcec3991fb514d20cd157ea38d1d88447601cbe4c9c98c591de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:53.271156Z","signature_b64":"umv+lbv7fyuDAplRgPlF9L/uHzJWtgeIWDSvFLlcvy/+FSvQco2rT7FzkpeEaU+SIUFxbH1X1JIB2SoWFN/SBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"277f1fccbea86285b6539e0a1a704f2d9424a38ffa6c199dbe2ac28583aa3d77","last_reissued_at":"2026-07-05T11:14:53.270642Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:53.270642Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UGPhysics: A Comprehensive Benchmark for Undergraduate Physics Reasoning with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Can Yang, Jiaxin Zhang, Qiyun Xu, Shizhe Diao, Tianhao Chen, Tong Xiao, Xin Xu, Yang Wang, Yuchen Yan","submitted_at":"2025-02-01T06:42:02Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in solving complex reasoning tasks, particularly in mathematics. However, the domain of physics reasoning presents unique challenges that have received significantly less attention. Existing benchmarks often fall short in evaluating LLMs' abilities on the breadth and depth of undergraduate-level physics, underscoring the need for a comprehensive evaluation. To fill this gap, we introduce UGPhysics, a large-scale and comprehensive benchmark specifically designed to evaluate UnderGraduate-level Physics (UGPhysics) reasoning w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00334","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00334/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.00334","created_at":"2026-07-05T11:14:53.270701+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.00334v4","created_at":"2026-07-05T11:14:53.270701+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00334","created_at":"2026-07-05T11:14:53.270701+00:00"},{"alias_kind":"pith_short_12","alias_value":"E57R7TF6VBRI","created_at":"2026-07-05T11:14:53.270701+00:00"},{"alias_kind":"pith_short_16","alias_value":"E57R7TF6VBRILNST","created_at":"2026-07-05T11:14:53.270701+00:00"},{"alias_kind":"pith_short_8","alias_value":"E57R7TF6","created_at":"2026-07-05T11:14:53.270701+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07962","citing_title":"ChronoPhyBench: Do MLLMs Truly Understand the World or Merely Exploit Language Priors?","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04978","citing_title":"Aligning Perception, Reasoning, Modeling and Interaction: A Survey on Physical AI","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18936","citing_title":"Fine-Tuning Small Reasoning Models for Quantum Field Theory","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08863","citing_title":"Hidden in Plain Sight: Visual-to-Symbolic Analytical Solution Inference from Field Visualizations","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW","json":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW.json","graph_json":"https://pith.science/api/pith-number/E57R7TF6VBRILNSTTYFBU4CPFW/graph.json","events_json":"https://pith.science/api/pith-number/E57R7TF6VBRILNSTTYFBU4CPFW/events.json","paper":"https://pith.science/paper/E57R7TF6"},"agent_actions":{"view_html":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW","download_json":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW.json","view_paper":"https://pith.science/paper/E57R7TF6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.00334&json=true","fetch_graph":"https://pith.science/api/pith-number/E57R7TF6VBRILNSTTYFBU4CPFW/graph.json","fetch_events":"https://pith.science/api/pith-number/E57R7TF6VBRILNSTTYFBU4CPFW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW/action/storage_attestation","attest_author":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW/action/author_attestation","sign_citation":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW/action/citation_signature","submit_replication":"https://pith.science/pith/E57R7TF6VBRILNSTTYFBU4CPFW/action/replication_record"}},"created_at":"2026-07-05T11:14:53.270701+00:00","updated_at":"2026-07-05T11:14:53.270701+00:00"}