{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TIGRJQ2NT4HIKHJSRILGYN2NQM","short_pith_number":"pith:TIGRJQ2N","schema_version":"1.0","canonical_sha256":"9a0d14c34d9f0e851d328a166c374d833f169f527291caa9e4357e61ad14da50","source":{"kind":"arxiv","id":"2508.21452","version":1},"attestation_state":"computed","paper":{"title":"From Canonical to Complex: Benchmarking LLM Capabilities in Undergraduate Thermodynamics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","physics.chem-ph"],"primary_cat":"physics.ed-ph","authors_text":"Anna Gei{\\ss}ler, Friedrich Sch\\\"oppler, Luca-Sophie Bien, Tobias Hertel","submitted_at":"2025-08-29T09:36:54Z","abstract_excerpt":"Large language models (LLMs) are increasingly considered as tutoring aids in science education. Yet their readiness for unsupervised use in undergraduate instruction remains uncertain, as reliable teaching requires more than fluent recall: it demands consistent, principle-grounded reasoning. Thermodynamics, with its compact laws and subtle distinctions between state and path functions, reversibility, and entropy, provides an ideal testbed for evaluating such capabilities. Here we present UTQA, a 50-item undergraduate thermodynamics question answering benchmark, covering ideal-gas processes, re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.21452","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"physics.ed-ph","submitted_at":"2025-08-29T09:36:54Z","cross_cats_sorted":["cs.CL","physics.chem-ph"],"title_canon_sha256":"dadddf728777ae1e70fb341f67ca3dd51daac686308afe3b583879c51b4c1ae4","abstract_canon_sha256":"62ad46f94ee7cb7dde466c6b3faa5820fb7e375d2a27ce0267d885bed8530faf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:35.058437Z","signature_b64":"qsuF2AIVCLv/SlhHNvE8xjjY/yiREU8hFwTTSk7qJ1y5PCT/XOBrsoKMs52ufa9ourjx1V8JEBF45KDQ/PgxAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a0d14c34d9f0e851d328a166c374d833f169f527291caa9e4357e61ad14da50","last_reissued_at":"2026-07-05T12:01:35.057900Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:35.057900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Canonical to Complex: Benchmarking LLM Capabilities in Undergraduate Thermodynamics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","physics.chem-ph"],"primary_cat":"physics.ed-ph","authors_text":"Anna Gei{\\ss}ler, Friedrich Sch\\\"oppler, Luca-Sophie Bien, Tobias Hertel","submitted_at":"2025-08-29T09:36:54Z","abstract_excerpt":"Large language models (LLMs) are increasingly considered as tutoring aids in science education. Yet their readiness for unsupervised use in undergraduate instruction remains uncertain, as reliable teaching requires more than fluent recall: it demands consistent, principle-grounded reasoning. Thermodynamics, with its compact laws and subtle distinctions between state and path functions, reversibility, and entropy, provides an ideal testbed for evaluating such capabilities. Here we present UTQA, a 50-item undergraduate thermodynamics question answering benchmark, covering ideal-gas processes, re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.21452","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.21452/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.21452","created_at":"2026-07-05T12:01:35.057961+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.21452v1","created_at":"2026-07-05T12:01:35.057961+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.21452","created_at":"2026-07-05T12:01:35.057961+00:00"},{"alias_kind":"pith_short_12","alias_value":"TIGRJQ2NT4HI","created_at":"2026-07-05T12:01:35.057961+00:00"},{"alias_kind":"pith_short_16","alias_value":"TIGRJQ2NT4HIKHJS","created_at":"2026-07-05T12:01:35.057961+00:00"},{"alias_kind":"pith_short_8","alias_value":"TIGRJQ2N","created_at":"2026-07-05T12:01:35.057961+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19758","citing_title":"ThermoQA: A Three-Tier Benchmark for Evaluating Thermodynamic Reasoning in Large Language Models","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM","json":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM.json","graph_json":"https://pith.science/api/pith-number/TIGRJQ2NT4HIKHJSRILGYN2NQM/graph.json","events_json":"https://pith.science/api/pith-number/TIGRJQ2NT4HIKHJSRILGYN2NQM/events.json","paper":"https://pith.science/paper/TIGRJQ2N"},"agent_actions":{"view_html":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM","download_json":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM.json","view_paper":"https://pith.science/paper/TIGRJQ2N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.21452&json=true","fetch_graph":"https://pith.science/api/pith-number/TIGRJQ2NT4HIKHJSRILGYN2NQM/graph.json","fetch_events":"https://pith.science/api/pith-number/TIGRJQ2NT4HIKHJSRILGYN2NQM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM/action/storage_attestation","attest_author":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM/action/author_attestation","sign_citation":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM/action/citation_signature","submit_replication":"https://pith.science/pith/TIGRJQ2NT4HIKHJSRILGYN2NQM/action/replication_record"}},"created_at":"2026-07-05T12:01:35.057961+00:00","updated_at":"2026-07-05T12:01:35.057961+00:00"}