{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FI4OHBYAUOFW7UY4SLFQHHME7J","short_pith_number":"pith:FI4OHBYA","schema_version":"1.0","canonical_sha256":"2a38e38700a38b6fd31c92cb039d84fa71fd42a6b0a41651178d3201dc9cc486","source":{"kind":"arxiv","id":"2402.08955","version":1},"attestation_state":"computed","paper":{"title":"Using Counterfactual Tasks to Evaluate the Generality of Analogical Reasoning in Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Martha Lewis, Melanie Mitchell","submitted_at":"2024-02-14T05:52:23Z","abstract_excerpt":"Large language models (LLMs) have performed well on several reasoning benchmarks, including ones that test analogical reasoning abilities. However, it has been debated whether they are actually performing humanlike abstract reasoning or instead employing less general processes that rely on similarity to what has been seen in their training data. Here we investigate the generality of analogy-making abilities previously claimed for LLMs (Webb, Holyoak, & Lu, 2023). We take one set of analogy problems used to evaluate LLMs and create a set of \"counterfactual\" variants-versions that test the same "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08955","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2024-02-14T05:52:23Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"6f78a54516808c95bfe1db95abab2465af48a74855662cf367e4a4ea8229f3f5","abstract_canon_sha256":"73ba70850044fdc1094dedc61b85d28bf1797d8cc25cf84c3fc001c3b0288ea4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:45:06.027838Z","signature_b64":"YQKgF6MHtOFYwCN5W7uChSs9IH17GZH5GHSLac/ZP4eVDQN/Y4N75KLiDKbTZpX5EDMwAfJ2sTTfnX1cDB21AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a38e38700a38b6fd31c92cb039d84fa71fd42a6b0a41651178d3201dc9cc486","last_reissued_at":"2026-07-05T07:45:06.027350Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:45:06.027350Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Using Counterfactual Tasks to Evaluate the Generality of Analogical Reasoning in Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Martha Lewis, Melanie Mitchell","submitted_at":"2024-02-14T05:52:23Z","abstract_excerpt":"Large language models (LLMs) have performed well on several reasoning benchmarks, including ones that test analogical reasoning abilities. However, it has been debated whether they are actually performing humanlike abstract reasoning or instead employing less general processes that rely on similarity to what has been seen in their training data. Here we investigate the generality of analogy-making abilities previously claimed for LLMs (Webb, Holyoak, & Lu, 2023). We take one set of analogy problems used to evaluate LLMs and create a set of \"counterfactual\" variants-versions that test the same "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08955","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08955/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08955","created_at":"2026-07-05T07:45:06.027410+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08955v1","created_at":"2026-07-05T07:45:06.027410+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08955","created_at":"2026-07-05T07:45:06.027410+00:00"},{"alias_kind":"pith_short_12","alias_value":"FI4OHBYAUOFW","created_at":"2026-07-05T07:45:06.027410+00:00"},{"alias_kind":"pith_short_16","alias_value":"FI4OHBYAUOFW7UY4","created_at":"2026-07-05T07:45:06.027410+00:00"},{"alias_kind":"pith_short_8","alias_value":"FI4OHBYA","created_at":"2026-07-05T07:45:06.027410+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21008","citing_title":"The Metanym Game: A Self-Contained, Self-Consistent LLM Peer-Community Benchmark for Structural Intelligence","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13607","citing_title":"Reasoning as Pattern Matching: Shared Mechanisms in Human and LLM Everyday Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12731","citing_title":"Normative Robustness as a Frontier for Non-Verifiable Reasoning in LLMs","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08230","citing_title":"Empowerment Gain and Causal Model Construction: Children and adults are sensitive to controllability and variability in their causal interventions","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01359","citing_title":"Structural Ranking of the Cognitive Plausibility of Computational Models of Analogy and Metaphors with the Minimal Cognitive Grid","ref_index":163,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J","json":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J.json","graph_json":"https://pith.science/api/pith-number/FI4OHBYAUOFW7UY4SLFQHHME7J/graph.json","events_json":"https://pith.science/api/pith-number/FI4OHBYAUOFW7UY4SLFQHHME7J/events.json","paper":"https://pith.science/paper/FI4OHBYA"},"agent_actions":{"view_html":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J","download_json":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J.json","view_paper":"https://pith.science/paper/FI4OHBYA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08955&json=true","fetch_graph":"https://pith.science/api/pith-number/FI4OHBYAUOFW7UY4SLFQHHME7J/graph.json","fetch_events":"https://pith.science/api/pith-number/FI4OHBYAUOFW7UY4SLFQHHME7J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J/action/storage_attestation","attest_author":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J/action/author_attestation","sign_citation":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J/action/citation_signature","submit_replication":"https://pith.science/pith/FI4OHBYAUOFW7UY4SLFQHHME7J/action/replication_record"}},"created_at":"2026-07-05T07:45:06.027410+00:00","updated_at":"2026-07-05T07:45:06.027410+00:00"}