{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FSC4S5AGUYKWHJLO6T35NJX5VF","short_pith_number":"pith:FSC4S5AG","schema_version":"1.0","canonical_sha256":"2c85c97406a61563a56ef4f7d6a6fda9439b66978904192d12f6b5dd118c8e45","source":{"kind":"arxiv","id":"2304.11164","version":1},"attestation_state":"computed","paper":{"title":"Dialectical language model evaluation: An initial appraisal of the commonsense spatial reasoning abilities of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anthony G Cohn, Jose Hernandez-Orallo","submitted_at":"2023-04-22T06:28:46Z","abstract_excerpt":"Language models have become very popular recently and many claims have been made about their abilities, including for commonsense reasoning. Given the increasingly better results of current language models on previous static benchmarks for commonsense reasoning, we explore an alternative dialectical evaluation. The goal of this kind of evaluation is not to obtain an aggregate performance value but to find failures and map the boundaries of the system. Dialoguing with the system gives the opportunity to check for consistency and get more reassurance of these boundaries beyond anecdotal evidence"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.11164","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-04-22T06:28:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c00e22a4a8a05d5ac9f7a8858d26a1b0aa16f2aea042c0fde7863417bfe076e4","abstract_canon_sha256":"bf0b41c6454705330453dc6d6bb1b129017f57bd3c23c53fbbce59d2a8ae2454"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:03:29.378152Z","signature_b64":"2ASHLAVXwJ9EUvgL+us7zZY0ydPBn+64+ynEsrYPV7aXRD6t5aNu7cwclyhEK+Sqa/JIB0Svs0gCNiQ75TMNBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c85c97406a61563a56ef4f7d6a6fda9439b66978904192d12f6b5dd118c8e45","last_reissued_at":"2026-07-05T06:03:29.377643Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:03:29.377643Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dialectical language model evaluation: An initial appraisal of the commonsense spatial reasoning abilities of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Anthony G Cohn, Jose Hernandez-Orallo","submitted_at":"2023-04-22T06:28:46Z","abstract_excerpt":"Language models have become very popular recently and many claims have been made about their abilities, including for commonsense reasoning. Given the increasingly better results of current language models on previous static benchmarks for commonsense reasoning, we explore an alternative dialectical evaluation. The goal of this kind of evaluation is not to obtain an aggregate performance value but to find failures and map the boundaries of the system. Dialoguing with the system gives the opportunity to check for consistency and get more reassurance of these boundaries beyond anecdotal evidence"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.11164","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.11164/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.11164","created_at":"2026-07-05T06:03:29.377718+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.11164v1","created_at":"2026-07-05T06:03:29.377718+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.11164","created_at":"2026-07-05T06:03:29.377718+00:00"},{"alias_kind":"pith_short_12","alias_value":"FSC4S5AGUYKW","created_at":"2026-07-05T06:03:29.377718+00:00"},{"alias_kind":"pith_short_16","alias_value":"FSC4S5AGUYKWHJLO","created_at":"2026-07-05T06:03:29.377718+00:00"},{"alias_kind":"pith_short_8","alias_value":"FSC4S5AG","created_at":"2026-07-05T06:03:29.377718+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31285","citing_title":"Spatial Reasoning via Modality Switching Between Language and Symbolic Representation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31285","citing_title":"Spatial Reasoning via Modality Switching Between Language and Symbolic Representation","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22811","citing_title":"GS-QA: A Benchmark for Geospatial Question Answering","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18380","citing_title":"QSTRBench: a New Benchmark to Evaluate the Ability of Language Models to Reason with Qualitative Spatial and Temporal Calculi","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF","json":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF.json","graph_json":"https://pith.science/api/pith-number/FSC4S5AGUYKWHJLO6T35NJX5VF/graph.json","events_json":"https://pith.science/api/pith-number/FSC4S5AGUYKWHJLO6T35NJX5VF/events.json","paper":"https://pith.science/paper/FSC4S5AG"},"agent_actions":{"view_html":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF","download_json":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF.json","view_paper":"https://pith.science/paper/FSC4S5AG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.11164&json=true","fetch_graph":"https://pith.science/api/pith-number/FSC4S5AGUYKWHJLO6T35NJX5VF/graph.json","fetch_events":"https://pith.science/api/pith-number/FSC4S5AGUYKWHJLO6T35NJX5VF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF/action/storage_attestation","attest_author":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF/action/author_attestation","sign_citation":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF/action/citation_signature","submit_replication":"https://pith.science/pith/FSC4S5AGUYKWHJLO6T35NJX5VF/action/replication_record"}},"created_at":"2026-07-05T06:03:29.377718+00:00","updated_at":"2026-07-05T06:03:29.377718+00:00"}