{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:7PVBANEOBZCWSMD27TS7OHDWED","short_pith_number":"pith:7PVBANEO","schema_version":"1.0","canonical_sha256":"fbea10348e0e4569307afce5f71c7620ed4e84f052509a382b1e27537d392c8d","source":{"kind":"arxiv","id":"2607.13899","version":1},"attestation_state":"computed","paper":{"title":"AIMO Interpretability Challenge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Adam Vawda-Oomerjee, Andreas Waldis, Barbara Plank, Chaoran Liu, Chuan Yang, Fazl Barez, Josef Kucha\\v{r}, Marek Kadl\\v{c}\\'ik, Michal Spiegel, Michal \\v{S}tef\\'anik, Philipp Mondorf, Pontus Stenetorp, Qianying Liu, Simon Frieder","submitted_at":"2026-07-15T14:41:28Z","abstract_excerpt":"We propose the AIMO Interpretability Challenge, a competition on distinguishing robust from spurious reasoning in frontier mathematical language models based on the models' internal mechanisms. The challenge is motivated by a central limitation of standard reasoning benchmarks: strong final-answer accuracy does not reveal whether a model relies on stable reasoning mechanisms or exploits brittle reasoning shortcuts. Building on AI Mathematical Olympiad (AIMO) problems and submissions, together with resources from the Fields Model Initiative, the competition will provide (1) newly-published olym"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.13899","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-15T14:41:28Z","cross_cats_sorted":[],"title_canon_sha256":"fe3fe107384cbff9da502432ac70277d92ecc7cc816680084ee3c2b53e25b33c","abstract_canon_sha256":"f5d576098c83929215c2edec22473947fd98aab31f1a506ad0770da538c38ea0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-16T01:23:12.493586Z","signature_b64":"RZLiaUQSJw5cIGuSoNnBcndwTHsxoSJwDeezxLuURDmGDUBUARCZIozMsQGGsy/CAnk2T3UoLg6vXmX1KuAfDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fbea10348e0e4569307afce5f71c7620ed4e84f052509a382b1e27537d392c8d","last_reissued_at":"2026-07-16T01:23:12.492734Z","signature_status":"signed_v1","first_computed_at":"2026-07-16T01:23:12.492734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AIMO Interpretability Challenge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Adam Vawda-Oomerjee, Andreas Waldis, Barbara Plank, Chaoran Liu, Chuan Yang, Fazl Barez, Josef Kucha\\v{r}, Marek Kadl\\v{c}\\'ik, Michal Spiegel, Michal \\v{S}tef\\'anik, Philipp Mondorf, Pontus Stenetorp, Qianying Liu, Simon Frieder","submitted_at":"2026-07-15T14:41:28Z","abstract_excerpt":"We propose the AIMO Interpretability Challenge, a competition on distinguishing robust from spurious reasoning in frontier mathematical language models based on the models' internal mechanisms. The challenge is motivated by a central limitation of standard reasoning benchmarks: strong final-answer accuracy does not reveal whether a model relies on stable reasoning mechanisms or exploits brittle reasoning shortcuts. Building on AI Mathematical Olympiad (AIMO) problems and submissions, together with resources from the Fields Model Initiative, the competition will provide (1) newly-published olym"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.13899","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.13899/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.13899","created_at":"2026-07-16T01:23:12.493168+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.13899v1","created_at":"2026-07-16T01:23:12.493168+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.13899","created_at":"2026-07-16T01:23:12.493168+00:00"},{"alias_kind":"pith_short_12","alias_value":"7PVBANEOBZCW","created_at":"2026-07-16T01:23:12.493168+00:00"},{"alias_kind":"pith_short_16","alias_value":"7PVBANEOBZCWSMD2","created_at":"2026-07-16T01:23:12.493168+00:00"},{"alias_kind":"pith_short_8","alias_value":"7PVBANEO","created_at":"2026-07-16T01:23:12.493168+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED","json":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED.json","graph_json":"https://pith.science/api/pith-number/7PVBANEOBZCWSMD27TS7OHDWED/graph.json","events_json":"https://pith.science/api/pith-number/7PVBANEOBZCWSMD27TS7OHDWED/events.json","paper":"https://pith.science/paper/7PVBANEO"},"agent_actions":{"view_html":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED","download_json":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED.json","view_paper":"https://pith.science/paper/7PVBANEO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.13899&json=true","fetch_graph":"https://pith.science/api/pith-number/7PVBANEOBZCWSMD27TS7OHDWED/graph.json","fetch_events":"https://pith.science/api/pith-number/7PVBANEOBZCWSMD27TS7OHDWED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED/action/storage_attestation","attest_author":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED/action/author_attestation","sign_citation":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED/action/citation_signature","submit_replication":"https://pith.science/pith/7PVBANEOBZCWSMD27TS7OHDWED/action/replication_record"}},"created_at":"2026-07-16T01:23:12.493168+00:00","updated_at":"2026-07-16T01:23:12.493168+00:00"}