{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DHQT7XYBBG6QCRL23EPT4ZU53S","short_pith_number":"pith:DHQT7XYB","schema_version":"1.0","canonical_sha256":"19e13fdf0109bd01457ad91f3e669ddca48740b849fdb58d91901cb05471f586","source":{"kind":"arxiv","id":"2210.15303","version":3},"attestation_state":"computed","paper":{"title":"Can language models handle recursively nested grammatical structures? A case study on comparing models and humans","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew Kyle Lampinen","submitted_at":"2022-10-27T10:25:12Z","abstract_excerpt":"How should we compare the capabilities of language models (LMs) and humans? I draw inspiration from comparative psychology to highlight some challenges. In particular, I consider a case study: processing of recursively nested grammatical structures. Prior work suggests that LMs cannot handle these structures as reliably as humans can. However, the humans were provided with instructions and training, while the LMs were evaluated zero-shot. I therefore match the evaluation more closely. Providing large LMs with a simple prompt -- substantially less content than the human training -- allows the L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.15303","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-27T10:25:12Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d75d4f547df4b0c776636aea86b48e6c238269a905acb5f016ec03237a3e8b7e","abstract_canon_sha256":"55b0a9537c3cac4f35c5c78efe4c76e3e98987315332aa6a9d47cdcb86ed528f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:42:28.660998Z","signature_b64":"hxSks1mK8WtApPS9U02DR7uN7QotAju+nKXzQHFw0ZRt+0dTLvizDDPP+KRHPDNrojJCxyQHln+FBSk7A7PrAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19e13fdf0109bd01457ad91f3e669ddca48740b849fdb58d91901cb05471f586","last_reissued_at":"2026-07-05T05:42:28.660520Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:42:28.660520Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can language models handle recursively nested grammatical structures? A case study on comparing models and humans","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Andrew Kyle Lampinen","submitted_at":"2022-10-27T10:25:12Z","abstract_excerpt":"How should we compare the capabilities of language models (LMs) and humans? I draw inspiration from comparative psychology to highlight some challenges. In particular, I consider a case study: processing of recursively nested grammatical structures. Prior work suggests that LMs cannot handle these structures as reliably as humans can. However, the humans were provided with instructions and training, while the LMs were evaluated zero-shot. I therefore match the evaluation more closely. Providing large LMs with a simple prompt -- substantially less content than the human training -- allows the L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.15303","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.15303/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.15303","created_at":"2026-07-05T05:42:28.660587+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.15303v3","created_at":"2026-07-05T05:42:28.660587+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.15303","created_at":"2026-07-05T05:42:28.660587+00:00"},{"alias_kind":"pith_short_12","alias_value":"DHQT7XYBBG6Q","created_at":"2026-07-05T05:42:28.660587+00:00"},{"alias_kind":"pith_short_16","alias_value":"DHQT7XYBBG6QCRL2","created_at":"2026-07-05T05:42:28.660587+00:00"},{"alias_kind":"pith_short_8","alias_value":"DHQT7XYB","created_at":"2026-07-05T05:42:28.660587+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.11061","citing_title":"Beyond Human-Like Processing: Large Language Models Perform Equivalently on Forward and Backward Scientific Text","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S","json":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S.json","graph_json":"https://pith.science/api/pith-number/DHQT7XYBBG6QCRL23EPT4ZU53S/graph.json","events_json":"https://pith.science/api/pith-number/DHQT7XYBBG6QCRL23EPT4ZU53S/events.json","paper":"https://pith.science/paper/DHQT7XYB"},"agent_actions":{"view_html":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S","download_json":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S.json","view_paper":"https://pith.science/paper/DHQT7XYB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.15303&json=true","fetch_graph":"https://pith.science/api/pith-number/DHQT7XYBBG6QCRL23EPT4ZU53S/graph.json","fetch_events":"https://pith.science/api/pith-number/DHQT7XYBBG6QCRL23EPT4ZU53S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S/action/storage_attestation","attest_author":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S/action/author_attestation","sign_citation":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S/action/citation_signature","submit_replication":"https://pith.science/pith/DHQT7XYBBG6QCRL23EPT4ZU53S/action/replication_record"}},"created_at":"2026-07-05T05:42:28.660587+00:00","updated_at":"2026-07-05T05:42:28.660587+00:00"}