{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SPS5PIG63ECBCXGQ5SDGIOUAIR","short_pith_number":"pith:SPS5PIG6","schema_version":"1.0","canonical_sha256":"93e5d7a0ded904115cd0ec86643a804452696ed7f3ee07c7afa9ec375df5b585","source":{"kind":"arxiv","id":"2305.19555","version":3},"attestation_state":"computed","paper":{"title":"Large Language Models Are Not Strong Abstract Reasoners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ga\\\"el Gendron, Gillian Dobbie, Michael Witbrock, Qiming Bao","submitted_at":"2023-05-31T04:50:29Z","abstract_excerpt":"Large Language Models have shown tremendous performance on a large variety of natural language processing tasks, ranging from text comprehension to common sense reasoning. However, the mechanisms responsible for this success remain opaque, and it is unclear whether LLMs can achieve human-like cognitive capabilities or whether these models are still fundamentally circumscribed. Abstract reasoning is a fundamental task for cognition, consisting of finding and applying a general pattern from few data. Evaluating deep neural architectures on this task could give insight into their potential limita"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.19555","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-31T04:50:29Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"5eae51d5cbe1e9dc791974725e48692c99202ab7a2fc4b423888cb75111ab82e","abstract_canon_sha256":"6877856794c3eb8da3ed4576783f445977917318fa85a6ab35ce500638613826"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:38.758937Z","signature_b64":"LEVcCYrbGqpkKHV+Mc3Bgc3XFEujIj97rxhcgzIzWrkwquFZF7YLjEYUJHDIOhWw5kCZ3hHXTCPTq0sfNV5qBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93e5d7a0ded904115cd0ec86643a804452696ed7f3ee07c7afa9ec375df5b585","last_reissued_at":"2026-07-05T07:29:38.758489Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:38.758489Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models Are Not Strong Abstract Reasoners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ga\\\"el Gendron, Gillian Dobbie, Michael Witbrock, Qiming Bao","submitted_at":"2023-05-31T04:50:29Z","abstract_excerpt":"Large Language Models have shown tremendous performance on a large variety of natural language processing tasks, ranging from text comprehension to common sense reasoning. However, the mechanisms responsible for this success remain opaque, and it is unclear whether LLMs can achieve human-like cognitive capabilities or whether these models are still fundamentally circumscribed. Abstract reasoning is a fundamental task for cognition, consisting of finding and applying a general pattern from few data. Evaluating deep neural architectures on this task could give insight into their potential limita"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.19555","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.19555/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.19555","created_at":"2026-07-05T07:29:38.758543+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.19555v3","created_at":"2026-07-05T07:29:38.758543+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.19555","created_at":"2026-07-05T07:29:38.758543+00:00"},{"alias_kind":"pith_short_12","alias_value":"SPS5PIG63ECB","created_at":"2026-07-05T07:29:38.758543+00:00"},{"alias_kind":"pith_short_16","alias_value":"SPS5PIG63ECBCXGQ","created_at":"2026-07-05T07:29:38.758543+00:00"},{"alias_kind":"pith_short_8","alias_value":"SPS5PIG6","created_at":"2026-07-05T07:29:38.758543+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26530","citing_title":"DiARC: Distinguishing Positive and Negative Samples Helps Improving ARC-like Reasoning Ability of Large Language Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23972","citing_title":"Why We Need World Models for AGI: Where LLMs Fail and How World Models May Outperform","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26530","citing_title":"DiARC: Distinguishing Positive and Negative Samples Helps Improving ARC-like Reasoning Ability of Large Language Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18907","citing_title":"Gradient-Based Program Synthesis with Neurally Interpreted Languages","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2402.01207","citing_title":"Efficient Causal Graph Discovery Using Large Language Models","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR","json":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR.json","graph_json":"https://pith.science/api/pith-number/SPS5PIG63ECBCXGQ5SDGIOUAIR/graph.json","events_json":"https://pith.science/api/pith-number/SPS5PIG63ECBCXGQ5SDGIOUAIR/events.json","paper":"https://pith.science/paper/SPS5PIG6"},"agent_actions":{"view_html":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR","download_json":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR.json","view_paper":"https://pith.science/paper/SPS5PIG6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.19555&json=true","fetch_graph":"https://pith.science/api/pith-number/SPS5PIG63ECBCXGQ5SDGIOUAIR/graph.json","fetch_events":"https://pith.science/api/pith-number/SPS5PIG63ECBCXGQ5SDGIOUAIR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR/action/storage_attestation","attest_author":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR/action/author_attestation","sign_citation":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR/action/citation_signature","submit_replication":"https://pith.science/pith/SPS5PIG63ECBCXGQ5SDGIOUAIR/action/replication_record"}},"created_at":"2026-07-05T07:29:38.758543+00:00","updated_at":"2026-07-05T07:29:38.758543+00:00"}