{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T72SP6LI3VI4AOUMGC6HT6YC2G","short_pith_number":"pith:T72SP6LI","schema_version":"1.0","canonical_sha256":"9ff527f968dd51c03a8c30bc79fb02d18be2b8a244f52743a0b15ee89a07ac82","source":{"kind":"arxiv","id":"2404.14883","version":3},"attestation_state":"computed","paper":{"title":"Language in Vivo vs. in Silico: Size Matters but Larger Language Models Still Do Not Comprehend Language on a Par with Humans Due to Impenetrable Semantic Reference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Evelina Leivada, Fritz Guenther, Vittoria Dentella","submitted_at":"2024-04-23T10:09:46Z","abstract_excerpt":"Understanding the limits of language is a prerequisite for Large Language Models (LLMs) to act as theories of natural language. LLM performance in some language tasks presents both quantitative and qualitative differences from that of humans, however it remains to be determined whether such differences are amenable to model size. This work investigates the critical role of model scaling, determining whether increases in size make up for such differences between humans and models. We test three LLMs from different families (Bard, 137 billion parameters; ChatGPT-3.5, 175 billion; ChatGPT-4, 1.5 "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14883","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-23T10:09:46Z","cross_cats_sorted":[],"title_canon_sha256":"fd3c60f935a279b9666fef31fdd0bfb7d3db5c141d251729466b497ed4dfe4ef","abstract_canon_sha256":"b0d2c975427cc2c9199537031312917e8707f15a1a30603af45cce5c0fcee65d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:10.119354Z","signature_b64":"2OeL+Mr0vkquX2SP5hY8vuf6+NiF/vb8R4JjYuWM9Zzmf3vUG96R9eQFLZhg65OOhAxgqH3wEVfjP7o9igN4Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ff527f968dd51c03a8c30bc79fb02d18be2b8a244f52743a0b15ee89a07ac82","last_reissued_at":"2026-07-05T11:28:10.118811Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:10.118811Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language in Vivo vs. in Silico: Size Matters but Larger Language Models Still Do Not Comprehend Language on a Par with Humans Due to Impenetrable Semantic Reference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Evelina Leivada, Fritz Guenther, Vittoria Dentella","submitted_at":"2024-04-23T10:09:46Z","abstract_excerpt":"Understanding the limits of language is a prerequisite for Large Language Models (LLMs) to act as theories of natural language. LLM performance in some language tasks presents both quantitative and qualitative differences from that of humans, however it remains to be determined whether such differences are amenable to model size. This work investigates the critical role of model scaling, determining whether increases in size make up for such differences between humans and models. We test three LLMs from different families (Bard, 137 billion parameters; ChatGPT-3.5, 175 billion; ChatGPT-4, 1.5 "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14883","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14883/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14883","created_at":"2026-07-05T11:28:10.118874+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14883v3","created_at":"2026-07-05T11:28:10.118874+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14883","created_at":"2026-07-05T11:28:10.118874+00:00"},{"alias_kind":"pith_short_12","alias_value":"T72SP6LI3VI4","created_at":"2026-07-05T11:28:10.118874+00:00"},{"alias_kind":"pith_short_16","alias_value":"T72SP6LI3VI4AOUM","created_at":"2026-07-05T11:28:10.118874+00:00"},{"alias_kind":"pith_short_8","alias_value":"T72SP6LI","created_at":"2026-07-05T11:28:10.118874+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.02225","citing_title":"Towards Fundamental Language Models: Does Linguistic Competence Scale with Model Size?","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G","json":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G.json","graph_json":"https://pith.science/api/pith-number/T72SP6LI3VI4AOUMGC6HT6YC2G/graph.json","events_json":"https://pith.science/api/pith-number/T72SP6LI3VI4AOUMGC6HT6YC2G/events.json","paper":"https://pith.science/paper/T72SP6LI"},"agent_actions":{"view_html":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G","download_json":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G.json","view_paper":"https://pith.science/paper/T72SP6LI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14883&json=true","fetch_graph":"https://pith.science/api/pith-number/T72SP6LI3VI4AOUMGC6HT6YC2G/graph.json","fetch_events":"https://pith.science/api/pith-number/T72SP6LI3VI4AOUMGC6HT6YC2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G/action/storage_attestation","attest_author":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G/action/author_attestation","sign_citation":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G/action/citation_signature","submit_replication":"https://pith.science/pith/T72SP6LI3VI4AOUMGC6HT6YC2G/action/replication_record"}},"created_at":"2026-07-05T11:28:10.118874+00:00","updated_at":"2026-07-05T11:28:10.118874+00:00"}