{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:K7YLIDBMHXUDWBW2P52OLWQFPU","short_pith_number":"pith:K7YLIDBM","schema_version":"1.0","canonical_sha256":"57f0b40c2c3de83b06da7f74e5da057d1f017a5d0e4858680b9c28b691ae120c","source":{"kind":"arxiv","id":"2505.16340","version":1},"attestation_state":"computed","paper":{"title":"Improving Chemical Understanding of LLMs via SMILES Parsing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jaehyung Kim, Sungsoo Ahn, Yunhui Jang","submitted_at":"2025-05-22T07:54:39Z","abstract_excerpt":"Large language models (LLMs) are increasingly recognized as powerful tools for scientific discovery, particularly in molecular science. A fundamental requirement for these models is the ability to accurately understand molecular structures, commonly encoded in the SMILES representation. However, current LLMs struggle to interpret SMILES, even failing to carry out basic tasks such as counting molecular rings. To address this limitation, we introduce CLEANMOL, a novel framework that formulates SMILES parsing into a suite of clean and deterministic tasks explicitly designed to promote graph-level"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16340","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-22T07:54:39Z","cross_cats_sorted":[],"title_canon_sha256":"f70f04feac8c71e3633e396d91ccc5ee3d2eeecd2dcfc4f4e1343dc64dfba83b","abstract_canon_sha256":"d40f53717719333728937ecc397b470d46ead453d6bbb89553d5e9e720dfbd93"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:34.005964Z","signature_b64":"yegVtGgveXTdjr0FqTk93qxXZTyUzathAJabxYnv0kPVwq8lWPYdDPhKnLUaHrnyqDHST3uOf7bWJPuRWYaQAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57f0b40c2c3de83b06da7f74e5da057d1f017a5d0e4858680b9c28b691ae120c","last_reissued_at":"2026-07-05T11:07:34.005481Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:34.005481Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Chemical Understanding of LLMs via SMILES Parsing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jaehyung Kim, Sungsoo Ahn, Yunhui Jang","submitted_at":"2025-05-22T07:54:39Z","abstract_excerpt":"Large language models (LLMs) are increasingly recognized as powerful tools for scientific discovery, particularly in molecular science. A fundamental requirement for these models is the ability to accurately understand molecular structures, commonly encoded in the SMILES representation. However, current LLMs struggle to interpret SMILES, even failing to carry out basic tasks such as counting molecular rings. To address this limitation, we introduce CLEANMOL, a novel framework that formulates SMILES parsing into a suite of clean and deterministic tasks explicitly designed to promote graph-level"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16340","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16340/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16340","created_at":"2026-07-05T11:07:34.005540+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16340v1","created_at":"2026-07-05T11:07:34.005540+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16340","created_at":"2026-07-05T11:07:34.005540+00:00"},{"alias_kind":"pith_short_12","alias_value":"K7YLIDBMHXUD","created_at":"2026-07-05T11:07:34.005540+00:00"},{"alias_kind":"pith_short_16","alias_value":"K7YLIDBMHXUDWBW2","created_at":"2026-07-05T11:07:34.005540+00:00"},{"alias_kind":"pith_short_8","alias_value":"K7YLIDBM","created_at":"2026-07-05T11:07:34.005540+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22926","citing_title":"Reaction-Network-Level Discovery of Ammonia Synthesis Catalysts via Ten-Million-Scale Generative Exploration","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU","json":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU.json","graph_json":"https://pith.science/api/pith-number/K7YLIDBMHXUDWBW2P52OLWQFPU/graph.json","events_json":"https://pith.science/api/pith-number/K7YLIDBMHXUDWBW2P52OLWQFPU/events.json","paper":"https://pith.science/paper/K7YLIDBM"},"agent_actions":{"view_html":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU","download_json":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU.json","view_paper":"https://pith.science/paper/K7YLIDBM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16340&json=true","fetch_graph":"https://pith.science/api/pith-number/K7YLIDBMHXUDWBW2P52OLWQFPU/graph.json","fetch_events":"https://pith.science/api/pith-number/K7YLIDBMHXUDWBW2P52OLWQFPU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU/action/storage_attestation","attest_author":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU/action/author_attestation","sign_citation":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU/action/citation_signature","submit_replication":"https://pith.science/pith/K7YLIDBMHXUDWBW2P52OLWQFPU/action/replication_record"}},"created_at":"2026-07-05T11:07:34.005540+00:00","updated_at":"2026-07-05T11:07:34.005540+00:00"}