{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HLNYJRRWQF4PO3TPXE75CQR4YK","short_pith_number":"pith:HLNYJRRW","schema_version":"1.0","canonical_sha256":"3adb84c6368178f76e6fb93fd1423cc28440a5dd64f38e5524b792a8ecc9a8b6","source":{"kind":"arxiv","id":"2312.11690","version":1},"attestation_state":"computed","paper":{"title":"Agent-based Learning of Materials Datasets from Scientific Literature","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Mehrad Ansari, Seyed Mohamad Moosavi","submitted_at":"2023-12-18T20:29:58Z","abstract_excerpt":"Advancements in machine learning and artificial intelligence are transforming materials discovery. Yet, the availability of structured experimental data remains a bottleneck. The vast corpus of scientific literature presents a valuable and rich resource of such data. However, manual dataset creation from these resources is challenging due to issues in maintaining quality and consistency, scalability limitations, and the risk of human error and bias. Therefore, in this work, we develop a chemist AI agent, powered by large language models (LLMs), to overcome these challenges by autonomously crea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.11690","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-12-18T20:29:58Z","cross_cats_sorted":[],"title_canon_sha256":"670878ffcee89f6fcb3fdb882e0322b17e51ee1fe016adfedfca050fa1492633","abstract_canon_sha256":"7de9f29371d060649edbee4904c840a3af35b3a22d730be84c45515fce719bdb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:25:52.957386Z","signature_b64":"jfZGar6ZM9Uij5DJh4AJ2JNu9fqNffLnnF6bdcqTjmcAuzGTz4j/CkXmQXFeX0jHFvw5kEZen0VDnI5paTk+Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3adb84c6368178f76e6fb93fd1423cc28440a5dd64f38e5524b792a8ecc9a8b6","last_reissued_at":"2026-07-05T07:25:52.956909Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:25:52.956909Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Agent-based Learning of Materials Datasets from Scientific Literature","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Mehrad Ansari, Seyed Mohamad Moosavi","submitted_at":"2023-12-18T20:29:58Z","abstract_excerpt":"Advancements in machine learning and artificial intelligence are transforming materials discovery. Yet, the availability of structured experimental data remains a bottleneck. The vast corpus of scientific literature presents a valuable and rich resource of such data. However, manual dataset creation from these resources is challenging due to issues in maintaining quality and consistency, scalability limitations, and the risk of human error and bias. Therefore, in this work, we develop a chemist AI agent, powered by large language models (LLMs), to overcome these challenges by autonomously crea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.11690","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.11690/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.11690","created_at":"2026-07-05T07:25:52.956968+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.11690v1","created_at":"2026-07-05T07:25:52.956968+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.11690","created_at":"2026-07-05T07:25:52.956968+00:00"},{"alias_kind":"pith_short_12","alias_value":"HLNYJRRWQF4P","created_at":"2026-07-05T07:25:52.956968+00:00"},{"alias_kind":"pith_short_16","alias_value":"HLNYJRRWQF4PO3TP","created_at":"2026-07-05T07:25:52.956968+00:00"},{"alias_kind":"pith_short_8","alias_value":"HLNYJRRW","created_at":"2026-07-05T07:25:52.956968+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.15221","citing_title":"Reflections from the 2024 Large Language Model (LLM) Hackathon for Applications in Materials Science and Chemistry","ref_index":113,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK","json":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK.json","graph_json":"https://pith.science/api/pith-number/HLNYJRRWQF4PO3TPXE75CQR4YK/graph.json","events_json":"https://pith.science/api/pith-number/HLNYJRRWQF4PO3TPXE75CQR4YK/events.json","paper":"https://pith.science/paper/HLNYJRRW"},"agent_actions":{"view_html":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK","download_json":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK.json","view_paper":"https://pith.science/paper/HLNYJRRW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.11690&json=true","fetch_graph":"https://pith.science/api/pith-number/HLNYJRRWQF4PO3TPXE75CQR4YK/graph.json","fetch_events":"https://pith.science/api/pith-number/HLNYJRRWQF4PO3TPXE75CQR4YK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK/action/storage_attestation","attest_author":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK/action/author_attestation","sign_citation":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK/action/citation_signature","submit_replication":"https://pith.science/pith/HLNYJRRWQF4PO3TPXE75CQR4YK/action/replication_record"}},"created_at":"2026-07-05T07:25:52.956968+00:00","updated_at":"2026-07-05T07:25:52.956968+00:00"}