{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2GKJHABR5BXXVULKTABQTKX64W","short_pith_number":"pith:2GKJHABR","schema_version":"1.0","canonical_sha256":"d194938031e86f7ad16a980309aafee592e2961fe7f12a4580c0ce18c53d5c66","source":{"kind":"arxiv","id":"2302.04752","version":2},"attestation_state":"computed","paper":{"title":"Benchmarks for Automated Commonsense Reasoning: A Survey","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ernest Davis","submitted_at":"2023-02-09T16:34:30Z","abstract_excerpt":"More than one hundred benchmarks have been developed to test the commonsense knowledge and commonsense reasoning abilities of artificial intelligence (AI) systems. However, these benchmarks are often flawed and many aspects of common sense remain untested. Consequently, we do not currently have any reliable way of measuring to what extent existing AI systems have achieved these abilities. This paper surveys the development and uses of AI commonsense benchmarks. We discuss the nature of common sense; the role of common sense in AI; the goals served by constructing commonsense benchmarks; and de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.04752","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2023-02-09T16:34:30Z","cross_cats_sorted":[],"title_canon_sha256":"ab5aebb6be5c6a1d30b0fea2701f3691b6b0b7a35e2f59b285ac999ece2f9b26","abstract_canon_sha256":"94f31e8e6cfaf3b783ccd1397bd09798fbb78ee544893b45f4b6edb06fb2b5f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:44:52.634199Z","signature_b64":"ahB8srVMyULxubystr7bzogNedeTqVkug7aO5Xaf3zuImQjUwYBZben0bIey18zgPLygbAP/h75fnbX2bN73Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d194938031e86f7ad16a980309aafee592e2961fe7f12a4580c0ce18c53d5c66","last_reissued_at":"2026-07-05T05:44:52.633713Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:44:52.633713Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarks for Automated Commonsense Reasoning: A Survey","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ernest Davis","submitted_at":"2023-02-09T16:34:30Z","abstract_excerpt":"More than one hundred benchmarks have been developed to test the commonsense knowledge and commonsense reasoning abilities of artificial intelligence (AI) systems. However, these benchmarks are often flawed and many aspects of common sense remain untested. Consequently, we do not currently have any reliable way of measuring to what extent existing AI systems have achieved these abilities. This paper surveys the development and uses of AI commonsense benchmarks. We discuss the nature of common sense; the role of common sense in AI; the goals served by constructing commonsense benchmarks; and de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.04752","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.04752/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.04752","created_at":"2026-07-05T05:44:52.633774+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.04752v2","created_at":"2026-07-05T05:44:52.633774+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.04752","created_at":"2026-07-05T05:44:52.633774+00:00"},{"alias_kind":"pith_short_12","alias_value":"2GKJHABR5BXX","created_at":"2026-07-05T05:44:52.633774+00:00"},{"alias_kind":"pith_short_16","alias_value":"2GKJHABR5BXXVULK","created_at":"2026-07-05T05:44:52.633774+00:00"},{"alias_kind":"pith_short_8","alias_value":"2GKJHABR","created_at":"2026-07-05T05:44:52.633774+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.09089","citing_title":"Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W","json":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W.json","graph_json":"https://pith.science/api/pith-number/2GKJHABR5BXXVULKTABQTKX64W/graph.json","events_json":"https://pith.science/api/pith-number/2GKJHABR5BXXVULKTABQTKX64W/events.json","paper":"https://pith.science/paper/2GKJHABR"},"agent_actions":{"view_html":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W","download_json":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W.json","view_paper":"https://pith.science/paper/2GKJHABR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.04752&json=true","fetch_graph":"https://pith.science/api/pith-number/2GKJHABR5BXXVULKTABQTKX64W/graph.json","fetch_events":"https://pith.science/api/pith-number/2GKJHABR5BXXVULKTABQTKX64W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W/action/storage_attestation","attest_author":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W/action/author_attestation","sign_citation":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W/action/citation_signature","submit_replication":"https://pith.science/pith/2GKJHABR5BXXVULKTABQTKX64W/action/replication_record"}},"created_at":"2026-07-05T05:44:52.633774+00:00","updated_at":"2026-07-05T05:44:52.633774+00:00"}