{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QCFM6J3Q2V2BWXZPQNEBIM5E5L","short_pith_number":"pith:QCFM6J3Q","schema_version":"1.0","canonical_sha256":"808acf2770d5741b5f2f83481433a4eadd14a1c23ce5785a99eae801a1bbfa4c","source":{"kind":"arxiv","id":"2411.14215","version":1},"attestation_state":"computed","paper":{"title":"Evaluating the Robustness of Analogical Reasoning in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Martha Lewis, Melanie Mitchell","submitted_at":"2024-11-21T15:25:08Z","abstract_excerpt":"LLMs have performed well on several reasoning benchmarks, including ones that test analogical reasoning abilities. However, there is debate on the extent to which they are performing general abstract reasoning versus employing non-robust processes, e.g., that overly rely on similarity to pre-training data. Here we investigate the robustness of analogy-making abilities previously claimed for LLMs on three of four domains studied by Webb, Holyoak, and Lu (2023): letter-string analogies, digit matrices, and story analogies. For each domain we test humans and GPT models on robustness to variants o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.14215","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-21T15:25:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d6b2e02cc3102505fb8abe2636c330ac647a09f4a2cb5407772ccae8aeaccee8","abstract_canon_sha256":"4819f2872a7f73efa7450561e9ffbcb62fb7c03d51b9de9c43bcbb676e326600"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:43.411685Z","signature_b64":"kKgrqcbKH1fgAkjHKz1KRcw4xqNazZqYJlTZdejLPBoi3ZV/PIWqY4JExw03+wEJFxgzA5ItLccNPu0wcv3zCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"808acf2770d5741b5f2f83481433a4eadd14a1c23ce5785a99eae801a1bbfa4c","last_reissued_at":"2026-07-05T09:38:43.411208Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:43.411208Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating the Robustness of Analogical Reasoning in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Martha Lewis, Melanie Mitchell","submitted_at":"2024-11-21T15:25:08Z","abstract_excerpt":"LLMs have performed well on several reasoning benchmarks, including ones that test analogical reasoning abilities. However, there is debate on the extent to which they are performing general abstract reasoning versus employing non-robust processes, e.g., that overly rely on similarity to pre-training data. Here we investigate the robustness of analogy-making abilities previously claimed for LLMs on three of four domains studied by Webb, Holyoak, and Lu (2023): letter-string analogies, digit matrices, and story analogies. For each domain we test humans and GPT models on robustness to variants o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14215","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.14215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.14215","created_at":"2026-07-05T09:38:43.411264+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.14215v1","created_at":"2026-07-05T09:38:43.411264+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14215","created_at":"2026-07-05T09:38:43.411264+00:00"},{"alias_kind":"pith_short_12","alias_value":"QCFM6J3Q2V2B","created_at":"2026-07-05T09:38:43.411264+00:00"},{"alias_kind":"pith_short_16","alias_value":"QCFM6J3Q2V2BWXZP","created_at":"2026-07-05T09:38:43.411264+00:00"},{"alias_kind":"pith_short_8","alias_value":"QCFM6J3Q","created_at":"2026-07-05T09:38:43.411264+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01152","citing_title":"AGC-Bench: Measuring Artificial General Creativity","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01152","citing_title":"AGC-Bench: Measuring Artificial General Creativity","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18852","citing_title":"Mechanistic Interpretability Needs Philosophy","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01359","citing_title":"Structural Ranking of the Cognitive Plausibility of Computational Models of Analogy and Metaphors with the Minimal Cognitive Grid","ref_index":225,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L","json":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L.json","graph_json":"https://pith.science/api/pith-number/QCFM6J3Q2V2BWXZPQNEBIM5E5L/graph.json","events_json":"https://pith.science/api/pith-number/QCFM6J3Q2V2BWXZPQNEBIM5E5L/events.json","paper":"https://pith.science/paper/QCFM6J3Q"},"agent_actions":{"view_html":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L","download_json":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L.json","view_paper":"https://pith.science/paper/QCFM6J3Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.14215&json=true","fetch_graph":"https://pith.science/api/pith-number/QCFM6J3Q2V2BWXZPQNEBIM5E5L/graph.json","fetch_events":"https://pith.science/api/pith-number/QCFM6J3Q2V2BWXZPQNEBIM5E5L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L/action/storage_attestation","attest_author":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L/action/author_attestation","sign_citation":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L/action/citation_signature","submit_replication":"https://pith.science/pith/QCFM6J3Q2V2BWXZPQNEBIM5E5L/action/replication_record"}},"created_at":"2026-07-05T09:38:43.411264+00:00","updated_at":"2026-07-05T09:38:43.411264+00:00"}