{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OJUVE6OCHLDVDI4ZY2MSUMV3YP","short_pith_number":"pith:OJUVE6OC","schema_version":"1.0","canonical_sha256":"72695279c23ac751a399c6992a32bbc3fbf6bd15dc36975000262a15b71b0125","source":{"kind":"arxiv","id":"2309.05452","version":2},"attestation_state":"computed","paper":{"title":"Evaluating the Deductive Competence of Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Spencer M. Seals, Valerie L. Shalin","submitted_at":"2023-09-11T13:47:07Z","abstract_excerpt":"The development of highly fluent large language models (LLMs) has prompted increased interest in assessing their reasoning and problem-solving capabilities. We investigate whether several LLMs can solve a classic type of deductive reasoning problem from the cognitive science literature. The tested LLMs have limited abilities to solve these problems in their conventional form. We performed follow up experiments to investigate if changes to the presentation format and content improve model performance. We do find performance differences between conditions; however, they do not improve overall pe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.05452","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-11T13:47:07Z","cross_cats_sorted":[],"title_canon_sha256":"cc48b2268bd953d1a571e4658688a5b22a2c19c8f266c4c641fe9f7dfd7bcfd2","abstract_canon_sha256":"3f06333acc5af32b5b7f3aebd8344d83a723ce3530e3f22b7a42becf90447c36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:58.072716Z","signature_b64":"XkfWdmlsrpiyM5gdR7KaBvkM/jxGQBfokj7Z6KykXWc74puHmsvO/oRNuEehaNBQVAlW8A7OuXmMAIocDGYvBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72695279c23ac751a399c6992a32bbc3fbf6bd15dc36975000262a15b71b0125","last_reissued_at":"2026-07-05T08:07:58.072302Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:58.072302Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating the Deductive Competence of Large Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Spencer M. Seals, Valerie L. Shalin","submitted_at":"2023-09-11T13:47:07Z","abstract_excerpt":"The development of highly fluent large language models (LLMs) has prompted increased interest in assessing their reasoning and problem-solving capabilities. We investigate whether several LLMs can solve a classic type of deductive reasoning problem from the cognitive science literature. The tested LLMs have limited abilities to solve these problems in their conventional form. We performed follow up experiments to investigate if changes to the presentation format and content improve model performance. We do find performance differences between conditions; however, they do not improve overall pe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.05452","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.05452/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.05452","created_at":"2026-07-05T08:07:58.072387+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.05452v2","created_at":"2026-07-05T08:07:58.072387+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.05452","created_at":"2026-07-05T08:07:58.072387+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJUVE6OCHLDV","created_at":"2026-07-05T08:07:58.072387+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJUVE6OCHLDVDI4Z","created_at":"2026-07-05T08:07:58.072387+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJUVE6OC","created_at":"2026-07-05T08:07:58.072387+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.11599","citing_title":"SR-FoT: A Syllogistic-Reasoning Framework of Thought for Large Language Models Tackling Knowledge-based Reasoning Tasks","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP","json":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP.json","graph_json":"https://pith.science/api/pith-number/OJUVE6OCHLDVDI4ZY2MSUMV3YP/graph.json","events_json":"https://pith.science/api/pith-number/OJUVE6OCHLDVDI4ZY2MSUMV3YP/events.json","paper":"https://pith.science/paper/OJUVE6OC"},"agent_actions":{"view_html":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP","download_json":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP.json","view_paper":"https://pith.science/paper/OJUVE6OC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.05452&json=true","fetch_graph":"https://pith.science/api/pith-number/OJUVE6OCHLDVDI4ZY2MSUMV3YP/graph.json","fetch_events":"https://pith.science/api/pith-number/OJUVE6OCHLDVDI4ZY2MSUMV3YP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP/action/storage_attestation","attest_author":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP/action/author_attestation","sign_citation":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP/action/citation_signature","submit_replication":"https://pith.science/pith/OJUVE6OCHLDVDI4ZY2MSUMV3YP/action/replication_record"}},"created_at":"2026-07-05T08:07:58.072387+00:00","updated_at":"2026-07-05T08:07:58.072387+00:00"}