{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G677YQ5KUMAMPH74UT3MW3LTBQ","short_pith_number":"pith:G677YQ5K","schema_version":"1.0","canonical_sha256":"37bffc43aaa300c79ffca4f6cb6d730c2004b33053274b3c2b306fe86b37ac08","source":{"kind":"arxiv","id":"2410.07114","version":5},"attestation_state":"computed","paper":{"title":"System 2 thinking in OpenAI's o1-preview model: Near-perfect performance on a mathematics exam","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CY","authors_text":"Dimitra Dodou, Joost de Winter, Yke Bauke Eisma","submitted_at":"2024-09-19T19:48:31Z","abstract_excerpt":"The processes underlying human cognition are often divided into System 1, which involves fast, intuitive thinking, and System 2, which involves slow, deliberate reasoning. Previously, large language models were criticized for lacking the deeper, more analytical capabilities of System 2. In September 2024, OpenAI introduced the o1 model series, designed to handle System 2-like reasoning. While OpenAI's benchmarks are promising, independent validation is still needed. In this study, we tested the o1-preview model twice on the Dutch 'Mathematics B' final exam. It scored a near-perfect 76 and 74 o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07114","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2024-09-19T19:48:31Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a23678f9df04e7d79a4859534864563b080ef3c7f67482a46469d70403e9304f","abstract_canon_sha256":"ae221bce209013d077754599dd9a14e914546e69d8d4299cd83eb73388882a0f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:47.790585Z","signature_b64":"m1Uw579uNX7+xOfADYGAwpAjyvpxPimWbPUjcqls1jM/uRv4bQvGupP+Wi7K5oBKl1chaus8DDEUgmmw7CPQCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"37bffc43aaa300c79ffca4f6cb6d730c2004b33053274b3c2b306fe86b37ac08","last_reissued_at":"2026-07-05T09:25:47.790068Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:47.790068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"System 2 thinking in OpenAI's o1-preview model: Near-perfect performance on a mathematics exam","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CY","authors_text":"Dimitra Dodou, Joost de Winter, Yke Bauke Eisma","submitted_at":"2024-09-19T19:48:31Z","abstract_excerpt":"The processes underlying human cognition are often divided into System 1, which involves fast, intuitive thinking, and System 2, which involves slow, deliberate reasoning. Previously, large language models were criticized for lacking the deeper, more analytical capabilities of System 2. In September 2024, OpenAI introduced the o1 model series, designed to handle System 2-like reasoning. While OpenAI's benchmarks are promising, independent validation is still needed. In this study, we tested the o1-preview model twice on the Dutch 'Mathematics B' final exam. It scored a near-perfect 76 and 74 o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07114","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07114","created_at":"2026-07-05T09:25:47.790136+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07114v5","created_at":"2026-07-05T09:25:47.790136+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07114","created_at":"2026-07-05T09:25:47.790136+00:00"},{"alias_kind":"pith_short_12","alias_value":"G677YQ5KUMAM","created_at":"2026-07-05T09:25:47.790136+00:00"},{"alias_kind":"pith_short_16","alias_value":"G677YQ5KUMAMPH74","created_at":"2026-07-05T09:25:47.790136+00:00"},{"alias_kind":"pith_short_8","alias_value":"G677YQ5K","created_at":"2026-07-05T09:25:47.790136+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ","json":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ.json","graph_json":"https://pith.science/api/pith-number/G677YQ5KUMAMPH74UT3MW3LTBQ/graph.json","events_json":"https://pith.science/api/pith-number/G677YQ5KUMAMPH74UT3MW3LTBQ/events.json","paper":"https://pith.science/paper/G677YQ5K"},"agent_actions":{"view_html":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ","download_json":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ.json","view_paper":"https://pith.science/paper/G677YQ5K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07114&json=true","fetch_graph":"https://pith.science/api/pith-number/G677YQ5KUMAMPH74UT3MW3LTBQ/graph.json","fetch_events":"https://pith.science/api/pith-number/G677YQ5KUMAMPH74UT3MW3LTBQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ/action/storage_attestation","attest_author":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ/action/author_attestation","sign_citation":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ/action/citation_signature","submit_replication":"https://pith.science/pith/G677YQ5KUMAMPH74UT3MW3LTBQ/action/replication_record"}},"created_at":"2026-07-05T09:25:47.790136+00:00","updated_at":"2026-07-05T09:25:47.790136+00:00"}