{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DUOLWIJLSXHO4DYFWZLUFSD4PI","short_pith_number":"pith:DUOLWIJL","schema_version":"1.0","canonical_sha256":"1d1cbb212b95ceee0f05b65742c87c7a142c400487c1ad70652023b639f2cafa","source":{"kind":"arxiv","id":"2311.09656","version":2},"attestation_state":"computed","paper":{"title":"Structured Chemistry Reasoning with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bing Yan, Jiawei Han, Lianhui Qin, Siru Ouyang, Xuan Liu, Yejin Choi, Zhuosheng Zhang","submitted_at":"2023-11-16T08:20:36Z","abstract_excerpt":"Large Language Models (LLMs) excel in diverse areas, yet struggle with complex scientific reasoning, especially in the field of chemistry. Different from the simple chemistry tasks (e.g., molecule classification) addressed in previous studies, complex chemistry problems require not only vast knowledge and precise calculation, but also compositional reasoning about rich dynamic interactions of different concepts (e.g., temperature changes). Our study shows that even advanced LLMs, like GPT-4, can fail easily in different ways. Interestingly, the errors often stem not from a lack of domain knowl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09656","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-16T08:20:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c85acab2fd04dcb60ecfe48edce56187ea426ad9a14173d7b69f097fcd13da02","abstract_canon_sha256":"c463eaff54f662d9e59e9439f943362a5aeac3207adfb777a66523a2397ec685"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:43:13.174983Z","signature_b64":"hzJBagkWWeiX0EKwgAIehhmJMpz+kHHkOEs72aemxcxm6E7hhRuy2soZq53fNXT4s6wWeL5nTT8o8u6YpUQEBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d1cbb212b95ceee0f05b65742c87c7a142c400487c1ad70652023b639f2cafa","last_reissued_at":"2026-07-05T07:43:13.174537Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:43:13.174537Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Structured Chemistry Reasoning with Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bing Yan, Jiawei Han, Lianhui Qin, Siru Ouyang, Xuan Liu, Yejin Choi, Zhuosheng Zhang","submitted_at":"2023-11-16T08:20:36Z","abstract_excerpt":"Large Language Models (LLMs) excel in diverse areas, yet struggle with complex scientific reasoning, especially in the field of chemistry. Different from the simple chemistry tasks (e.g., molecule classification) addressed in previous studies, complex chemistry problems require not only vast knowledge and precise calculation, but also compositional reasoning about rich dynamic interactions of different concepts (e.g., temperature changes). Our study shows that even advanced LLMs, like GPT-4, can fail easily in different ways. Interestingly, the errors often stem not from a lack of domain knowl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09656","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09656","created_at":"2026-07-05T07:43:13.174586+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09656v2","created_at":"2026-07-05T07:43:13.174586+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09656","created_at":"2026-07-05T07:43:13.174586+00:00"},{"alias_kind":"pith_short_12","alias_value":"DUOLWIJLSXHO","created_at":"2026-07-05T07:43:13.174586+00:00"},{"alias_kind":"pith_short_16","alias_value":"DUOLWIJLSXHO4DYF","created_at":"2026-07-05T07:43:13.174586+00:00"},{"alias_kind":"pith_short_8","alias_value":"DUOLWIJL","created_at":"2026-07-05T07:43:13.174586+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI","json":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI.json","graph_json":"https://pith.science/api/pith-number/DUOLWIJLSXHO4DYFWZLUFSD4PI/graph.json","events_json":"https://pith.science/api/pith-number/DUOLWIJLSXHO4DYFWZLUFSD4PI/events.json","paper":"https://pith.science/paper/DUOLWIJL"},"agent_actions":{"view_html":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI","download_json":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI.json","view_paper":"https://pith.science/paper/DUOLWIJL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09656&json=true","fetch_graph":"https://pith.science/api/pith-number/DUOLWIJLSXHO4DYFWZLUFSD4PI/graph.json","fetch_events":"https://pith.science/api/pith-number/DUOLWIJLSXHO4DYFWZLUFSD4PI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI/action/storage_attestation","attest_author":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI/action/author_attestation","sign_citation":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI/action/citation_signature","submit_replication":"https://pith.science/pith/DUOLWIJLSXHO4DYFWZLUFSD4PI/action/replication_record"}},"created_at":"2026-07-05T07:43:13.174586+00:00","updated_at":"2026-07-05T07:43:13.174586+00:00"}