{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BEWWY3K5KAKIUXYWHFHLYLWUHR","short_pith_number":"pith:BEWWY3K5","schema_version":"1.0","canonical_sha256":"092d6c6d5d50148a5f16394ebc2ed43c598af5ecb48c5f70121ac57692871535","source":{"kind":"arxiv","id":"2309.15129","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Cognitive Maps and Planning in Large Language Models with CogEval","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Felipe Vieira, Hamid Palangi, Hiteshi Sharma, Hosein Hasanbeig, Ida Momennejad, Jonathan Larson, Nebojsa Jojic, Robert Osazuwa Ness","submitted_at":"2023-09-25T01:20:13Z","abstract_excerpt":"Recently an influx of studies claim emergent cognitive abilities in large language models (LLMs). Yet, most rely on anecdotes, overlook contamination of training sets, or lack systematic Evaluation involving multiple tasks, control conditions, multiple iterations, and statistical robustness tests. Here we make two major contributions. First, we propose CogEval, a cognitive science-inspired protocol for the systematic evaluation of cognitive capacities in Large Language Models. The CogEval protocol can be followed for the evaluation of various abilities. Second, here we follow CogEval to system"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.15129","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2023-09-25T01:20:13Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"8ebcad3c652c2b4497aecec7120e9e8bafafb7156e7400f63b7d0035d8a2df57","abstract_canon_sha256":"7f6f87fc41675a8ffacd6e3595de31fb71300b20dfe4f0584db9a6b8ced9646c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:54:41.643836Z","signature_b64":"pE/A8hVvwuXcH73Bikdw/IKlEp5hRLGySOu3d6gBmd8dRLIpIIrgUXg6bjCjvP1qVavkyIBxzqNZqpwdJ/BOBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"092d6c6d5d50148a5f16394ebc2ed43c598af5ecb48c5f70121ac57692871535","last_reissued_at":"2026-07-05T06:54:41.643355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:54:41.643355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Cognitive Maps and Planning in Large Language Models with CogEval","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Felipe Vieira, Hamid Palangi, Hiteshi Sharma, Hosein Hasanbeig, Ida Momennejad, Jonathan Larson, Nebojsa Jojic, Robert Osazuwa Ness","submitted_at":"2023-09-25T01:20:13Z","abstract_excerpt":"Recently an influx of studies claim emergent cognitive abilities in large language models (LLMs). Yet, most rely on anecdotes, overlook contamination of training sets, or lack systematic Evaluation involving multiple tasks, control conditions, multiple iterations, and statistical robustness tests. Here we make two major contributions. First, we propose CogEval, a cognitive science-inspired protocol for the systematic evaluation of cognitive capacities in Large Language Models. The CogEval protocol can be followed for the evaluation of various abilities. Second, here we follow CogEval to system"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.15129","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.15129/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.15129","created_at":"2026-07-05T06:54:41.643414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.15129v1","created_at":"2026-07-05T06:54:41.643414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.15129","created_at":"2026-07-05T06:54:41.643414+00:00"},{"alias_kind":"pith_short_12","alias_value":"BEWWY3K5KAKI","created_at":"2026-07-05T06:54:41.643414+00:00"},{"alias_kind":"pith_short_16","alias_value":"BEWWY3K5KAKIUXYW","created_at":"2026-07-05T06:54:41.643414+00:00"},{"alias_kind":"pith_short_8","alias_value":"BEWWY3K5","created_at":"2026-07-05T06:54:41.643414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07544","citing_title":"Position: We Need An Algorithmic Understanding of Generative AI","ref_index":2020,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR","json":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR.json","graph_json":"https://pith.science/api/pith-number/BEWWY3K5KAKIUXYWHFHLYLWUHR/graph.json","events_json":"https://pith.science/api/pith-number/BEWWY3K5KAKIUXYWHFHLYLWUHR/events.json","paper":"https://pith.science/paper/BEWWY3K5"},"agent_actions":{"view_html":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR","download_json":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR.json","view_paper":"https://pith.science/paper/BEWWY3K5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.15129&json=true","fetch_graph":"https://pith.science/api/pith-number/BEWWY3K5KAKIUXYWHFHLYLWUHR/graph.json","fetch_events":"https://pith.science/api/pith-number/BEWWY3K5KAKIUXYWHFHLYLWUHR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR/action/storage_attestation","attest_author":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR/action/author_attestation","sign_citation":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR/action/citation_signature","submit_replication":"https://pith.science/pith/BEWWY3K5KAKIUXYWHFHLYLWUHR/action/replication_record"}},"created_at":"2026-07-05T06:54:41.643414+00:00","updated_at":"2026-07-05T06:54:41.643414+00:00"}