{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:T5H7IUTSM7TOTU2GCZGRJZHGFW","short_pith_number":"pith:T5H7IUTS","schema_version":"1.0","canonical_sha256":"9f4ff4527267e6e9d346164d14e4e62d90214138bba1d9f4bf70603211c99387","source":{"kind":"arxiv","id":"2306.04610","version":1},"attestation_state":"computed","paper":{"title":"The Two Word Test: A Semantic Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Nicholas Riccardi, Rutvik H. Desai","submitted_at":"2023-06-07T17:22:03Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable abilities recently, including passing advanced professional exams and demanding benchmark tests. This performance has led many to suggest that they are close to achieving humanlike or 'true' understanding of language, and even Artificial General Intelligence (AGI). Here, we provide a new open-source benchmark that can assess semantic abilities of LLMs using two-word phrases using a task that can be performed relatively easily by humans without advanced training. Combining multiple words into a single concept is a fundamental aspect of human la"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.04610","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-07T17:22:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b66e787a5d71458b1f5338f4746fbbe94337d5f4aa73b53e0b3c30b1519e8833","abstract_canon_sha256":"af40c6cca65eacb80e8983a0156e3417221f543171773ba3d78c2f7097b13b78"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:18:33.047589Z","signature_b64":"QvHSUAipVjH5sFDRKp0oK7l2rtSzk9zDeRrTwmNz+58JQhj0yXSHF/Rsp19kzMLYjvqmtQnC38Ekp0tmfCZ8BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f4ff4527267e6e9d346164d14e4e62d90214138bba1d9f4bf70603211c99387","last_reissued_at":"2026-07-05T06:18:33.047173Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:18:33.047173Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Two Word Test: A Semantic Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Nicholas Riccardi, Rutvik H. Desai","submitted_at":"2023-06-07T17:22:03Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable abilities recently, including passing advanced professional exams and demanding benchmark tests. This performance has led many to suggest that they are close to achieving humanlike or 'true' understanding of language, and even Artificial General Intelligence (AGI). Here, we provide a new open-source benchmark that can assess semantic abilities of LLMs using two-word phrases using a task that can be performed relatively easily by humans without advanced training. Combining multiple words into a single concept is a fundamental aspect of human la"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.04610","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.04610/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.04610","created_at":"2026-07-05T06:18:33.047226+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.04610v1","created_at":"2026-07-05T06:18:33.047226+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.04610","created_at":"2026-07-05T06:18:33.047226+00:00"},{"alias_kind":"pith_short_12","alias_value":"T5H7IUTSM7TO","created_at":"2026-07-05T06:18:33.047226+00:00"},{"alias_kind":"pith_short_16","alias_value":"T5H7IUTSM7TOTU2G","created_at":"2026-07-05T06:18:33.047226+00:00"},{"alias_kind":"pith_short_8","alias_value":"T5H7IUTS","created_at":"2026-07-05T06:18:33.047226+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.15922","citing_title":"Language Models to Support Multi-Label Classification of Industrial Data","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW","json":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW.json","graph_json":"https://pith.science/api/pith-number/T5H7IUTSM7TOTU2GCZGRJZHGFW/graph.json","events_json":"https://pith.science/api/pith-number/T5H7IUTSM7TOTU2GCZGRJZHGFW/events.json","paper":"https://pith.science/paper/T5H7IUTS"},"agent_actions":{"view_html":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW","download_json":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW.json","view_paper":"https://pith.science/paper/T5H7IUTS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.04610&json=true","fetch_graph":"https://pith.science/api/pith-number/T5H7IUTSM7TOTU2GCZGRJZHGFW/graph.json","fetch_events":"https://pith.science/api/pith-number/T5H7IUTSM7TOTU2GCZGRJZHGFW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW/action/storage_attestation","attest_author":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW/action/author_attestation","sign_citation":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW/action/citation_signature","submit_replication":"https://pith.science/pith/T5H7IUTSM7TOTU2GCZGRJZHGFW/action/replication_record"}},"created_at":"2026-07-05T06:18:33.047226+00:00","updated_at":"2026-07-05T06:18:33.047226+00:00"}