{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2014:GLFHFKSGOXV4INONCUP5RCHOMK","short_pith_number":"pith:GLFHFKSG","schema_version":"1.0","canonical_sha256":"32ca72aa4675ebc435cd151fd888ee62889ef9cb314fe61d9e0b15b0e39b77e2","source":{"kind":"arxiv","id":"1411.1629","version":2},"attestation_state":"computed","paper":{"title":"The Limitations of Standardized Science Tests as Benchmarks for Artificial Intelligence Research: Position Paper","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ernest Davis","submitted_at":"2014-11-06T14:44:12Z","abstract_excerpt":"In this position paper, I argue that standardized tests for elementary science such as SAT or Regents tests are not very good benchmarks for measuring the progress of artificial intelligence systems in understanding basic science. The primary problem is that these tests are designed to test aspects of knowledge and ability that are challenging for people; the aspects that are challenging for AI systems are very different. In particular, standardized tests do not test knowledge that is obvious for people; none of this knowledge can be assumed in AI systems. Individual standardized tests also ha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1411.1629","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2014-11-06T14:44:12Z","cross_cats_sorted":[],"title_canon_sha256":"b34b37161eabb4290d9ee34152ba61a80b66b80f687fdc676cafe3046c5fc54b","abstract_canon_sha256":"74842b9de08fc551f394b9340ec5627c65f57d237ed817af1755c2f9d8f92c89"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T01:29:59.261462Z","signature_b64":"ff1Kdm1TrGizAJz/s4Lel3KgcdHrKsyx+H3mJZaLLhdoBtsmkMHAehi4b0ZzjqXaLs3Taa2SOIzJ0f2DFoVOAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"32ca72aa4675ebc435cd151fd888ee62889ef9cb314fe61d9e0b15b0e39b77e2","last_reissued_at":"2026-05-18T01:29:59.260999Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T01:29:59.260999Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Limitations of Standardized Science Tests as Benchmarks for Artificial Intelligence Research: Position Paper","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ernest Davis","submitted_at":"2014-11-06T14:44:12Z","abstract_excerpt":"In this position paper, I argue that standardized tests for elementary science such as SAT or Regents tests are not very good benchmarks for measuring the progress of artificial intelligence systems in understanding basic science. The primary problem is that these tests are designed to test aspects of knowledge and ability that are challenging for people; the aspects that are challenging for AI systems are very different. In particular, standardized tests do not test knowledge that is obvious for people; none of this knowledge can be assumed in AI systems. Individual standardized tests also ha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1411.1629","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1411.1629","created_at":"2026-05-18T01:29:59.261068+00:00"},{"alias_kind":"arxiv_version","alias_value":"1411.1629v2","created_at":"2026-05-18T01:29:59.261068+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1411.1629","created_at":"2026-05-18T01:29:59.261068+00:00"},{"alias_kind":"pith_short_12","alias_value":"GLFHFKSGOXV4","created_at":"2026-05-18T12:28:30.664211+00:00"},{"alias_kind":"pith_short_16","alias_value":"GLFHFKSGOXV4INON","created_at":"2026-05-18T12:28:30.664211+00:00"},{"alias_kind":"pith_short_8","alias_value":"GLFHFKSG","created_at":"2026-05-18T12:28:30.664211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2511.05501","citing_title":"Towards Real-World Validity in Generative AI Benchmarks: Understanding and Designing Domain-Centered Evaluations for Journalism Practitioners","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK","json":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK.json","graph_json":"https://pith.science/api/pith-number/GLFHFKSGOXV4INONCUP5RCHOMK/graph.json","events_json":"https://pith.science/api/pith-number/GLFHFKSGOXV4INONCUP5RCHOMK/events.json","paper":"https://pith.science/paper/GLFHFKSG"},"agent_actions":{"view_html":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK","download_json":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK.json","view_paper":"https://pith.science/paper/GLFHFKSG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1411.1629&json=true","fetch_graph":"https://pith.science/api/pith-number/GLFHFKSGOXV4INONCUP5RCHOMK/graph.json","fetch_events":"https://pith.science/api/pith-number/GLFHFKSGOXV4INONCUP5RCHOMK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK/action/storage_attestation","attest_author":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK/action/author_attestation","sign_citation":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK/action/citation_signature","submit_replication":"https://pith.science/pith/GLFHFKSGOXV4INONCUP5RCHOMK/action/replication_record"}},"created_at":"2026-05-18T01:29:59.261068+00:00","updated_at":"2026-05-18T01:29:59.261068+00:00"}