{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2WDKVYLUZ4WXLCTRGD3RWZ44DA","short_pith_number":"pith:2WDKVYLU","schema_version":"1.0","canonical_sha256":"d586aae174cf2d758a7130f71b679c18146ee65adacbefc629504188a885de12","source":{"kind":"arxiv","id":"2308.09115","version":1},"attestation_state":"computed","paper":{"title":"MaScQA: A Question Answering Dataset for Investigating Materials Science Knowledge of Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cond-mat.mtrl-sci"],"primary_cat":"cs.CL","authors_text":"Jayadeva, Mausam, Mohd Zaki, N. M. Anoop Krishnan","submitted_at":"2023-08-17T17:51:05Z","abstract_excerpt":"Information extraction and textual comprehension from materials literature are vital for developing an exhaustive knowledge base that enables accelerated materials discovery. Language models have demonstrated their capability to answer domain-specific questions and retrieve information from knowledge bases. However, there are no benchmark datasets in the materials domain that can evaluate the understanding of the key concepts by these language models. In this work, we curate a dataset of 650 challenging questions from the materials domain that require the knowledge and skills of a materials st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.09115","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-17T17:51:05Z","cross_cats_sorted":["cond-mat.mtrl-sci"],"title_canon_sha256":"a6520a829ebfb279d1e9543ab4fbd77c36f9f43e5638e6e2204fb4f000ef8ec6","abstract_canon_sha256":"2a031fdc7c0ba34bd718547e13de6eaf43a53aa706f10865bb46f6b7a1b30e4d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:42:27.838962Z","signature_b64":"LDn67T/l/hp8oMuqC3WO/BMx90EMLFPmhXBfBNW3u3b7FUUOS2P2Njnc2Achw7DunTkDDwi7APmxIqT4pjVoCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d586aae174cf2d758a7130f71b679c18146ee65adacbefc629504188a885de12","last_reissued_at":"2026-07-05T06:42:27.838474Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:42:27.838474Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MaScQA: A Question Answering Dataset for Investigating Materials Science Knowledge of Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cond-mat.mtrl-sci"],"primary_cat":"cs.CL","authors_text":"Jayadeva, Mausam, Mohd Zaki, N. M. Anoop Krishnan","submitted_at":"2023-08-17T17:51:05Z","abstract_excerpt":"Information extraction and textual comprehension from materials literature are vital for developing an exhaustive knowledge base that enables accelerated materials discovery. Language models have demonstrated their capability to answer domain-specific questions and retrieve information from knowledge bases. However, there are no benchmark datasets in the materials domain that can evaluate the understanding of the key concepts by these language models. In this work, we curate a dataset of 650 challenging questions from the materials domain that require the knowledge and skills of a materials st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.09115","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.09115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.09115","created_at":"2026-07-05T06:42:27.838534+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.09115v1","created_at":"2026-07-05T06:42:27.838534+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.09115","created_at":"2026-07-05T06:42:27.838534+00:00"},{"alias_kind":"pith_short_12","alias_value":"2WDKVYLUZ4WX","created_at":"2026-07-05T06:42:27.838534+00:00"},{"alias_kind":"pith_short_16","alias_value":"2WDKVYLUZ4WXLCTR","created_at":"2026-07-05T06:42:27.838534+00:00"},{"alias_kind":"pith_short_8","alias_value":"2WDKVYLU","created_at":"2026-07-05T06:42:27.838534+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.11229","citing_title":"RECIPER: A Dual-View Retrieval Pipeline for Procedure-Oriented Materials Question Answering","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA","json":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA.json","graph_json":"https://pith.science/api/pith-number/2WDKVYLUZ4WXLCTRGD3RWZ44DA/graph.json","events_json":"https://pith.science/api/pith-number/2WDKVYLUZ4WXLCTRGD3RWZ44DA/events.json","paper":"https://pith.science/paper/2WDKVYLU"},"agent_actions":{"view_html":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA","download_json":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA.json","view_paper":"https://pith.science/paper/2WDKVYLU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.09115&json=true","fetch_graph":"https://pith.science/api/pith-number/2WDKVYLUZ4WXLCTRGD3RWZ44DA/graph.json","fetch_events":"https://pith.science/api/pith-number/2WDKVYLUZ4WXLCTRGD3RWZ44DA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA/action/storage_attestation","attest_author":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA/action/author_attestation","sign_citation":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA/action/citation_signature","submit_replication":"https://pith.science/pith/2WDKVYLUZ4WXLCTRGD3RWZ44DA/action/replication_record"}},"created_at":"2026-07-05T06:42:27.838534+00:00","updated_at":"2026-07-05T06:42:27.838534+00:00"}