{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NYHYQYTDMNFPZCE3N6TKET2FS3","short_pith_number":"pith:NYHYQYTD","schema_version":"1.0","canonical_sha256":"6e0f886263634afc889b6fa6a24f4596cac57cbbd740912c2a6e95ade0093d09","source":{"kind":"arxiv","id":"2502.13295","version":3},"attestation_state":"computed","paper":{"title":"Demonstrating specification gaming in reasoning models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alexander Bondarenko, Denis Volk, Dmitrii Volkov, Jeffrey Ladish","submitted_at":"2025-02-18T21:32:24Z","abstract_excerpt":"We demonstrate LLM agent specification gaming by instructing models to win against a chess engine. We find reasoning models like OpenAI o3 and DeepSeek R1 will often hack the benchmark by default, while language models like GPT-4o and Claude 3.5 Sonnet need to be told that normal play won't work to hack.\n  We improve upon prior work like (Hubinger et al., 2024; Meinke et al., 2024; Weij et al., 2024) by using realistic task prompts and avoiding excess nudging. Our results suggest reasoning models may resort to hacking to solve difficult problems, as observed in OpenAI (2024)'s o1 Docker escape"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.13295","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-18T21:32:24Z","cross_cats_sorted":[],"title_canon_sha256":"cffdfdb79869fe4d25496b2abe5789bbbbdddca143146a8c34f93b27d5734606","abstract_canon_sha256":"e22f45aa9845c4874fe3249401a1affb4bedbfdc8ee3a6f7949034af64cdabb1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:57.472457Z","signature_b64":"AG3apX/Lnkxy/jlx1u8Q7UhIrXkHdoJT1u3nOTSBoLB5U5n91CK0FtckMMNeVUCx0TDe1fR505LNsITta0fPBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e0f886263634afc889b6fa6a24f4596cac57cbbd740912c2a6e95ade0093d09","last_reissued_at":"2026-07-05T11:59:57.471964Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:57.471964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Demonstrating specification gaming in reasoning models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alexander Bondarenko, Denis Volk, Dmitrii Volkov, Jeffrey Ladish","submitted_at":"2025-02-18T21:32:24Z","abstract_excerpt":"We demonstrate LLM agent specification gaming by instructing models to win against a chess engine. We find reasoning models like OpenAI o3 and DeepSeek R1 will often hack the benchmark by default, while language models like GPT-4o and Claude 3.5 Sonnet need to be told that normal play won't work to hack.\n  We improve upon prior work like (Hubinger et al., 2024; Meinke et al., 2024; Weij et al., 2024) by using realistic task prompts and avoiding excess nudging. Our results suggest reasoning models may resort to hacking to solve difficult problems, as observed in OpenAI (2024)'s o1 Docker escape"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13295","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.13295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.13295","created_at":"2026-07-05T11:59:57.472024+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.13295v3","created_at":"2026-07-05T11:59:57.472024+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13295","created_at":"2026-07-05T11:59:57.472024+00:00"},{"alias_kind":"pith_short_12","alias_value":"NYHYQYTDMNFP","created_at":"2026-07-05T11:59:57.472024+00:00"},{"alias_kind":"pith_short_16","alias_value":"NYHYQYTDMNFPZCE3","created_at":"2026-07-05T11:59:57.472024+00:00"},{"alias_kind":"pith_short_8","alias_value":"NYHYQYTD","created_at":"2026-07-05T11:59:57.472024+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00038","citing_title":"Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28863","citing_title":"Defeat Devices in AI Systems","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20744","citing_title":"Hack-Verifiable Environments: Towards Evaluating Reward Hacking at Scale","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23338","citing_title":"A Systematic Survey of Security Threats and Defenses in LLM-Based AI Agents: A Layered Attack Surface Framework","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00226","citing_title":"Why Do LLMs Struggle in Strategic Play? Broken Links Between Observations, Beliefs, and Actions","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3","json":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3.json","graph_json":"https://pith.science/api/pith-number/NYHYQYTDMNFPZCE3N6TKET2FS3/graph.json","events_json":"https://pith.science/api/pith-number/NYHYQYTDMNFPZCE3N6TKET2FS3/events.json","paper":"https://pith.science/paper/NYHYQYTD"},"agent_actions":{"view_html":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3","download_json":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3.json","view_paper":"https://pith.science/paper/NYHYQYTD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.13295&json=true","fetch_graph":"https://pith.science/api/pith-number/NYHYQYTDMNFPZCE3N6TKET2FS3/graph.json","fetch_events":"https://pith.science/api/pith-number/NYHYQYTDMNFPZCE3N6TKET2FS3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3/action/storage_attestation","attest_author":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3/action/author_attestation","sign_citation":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3/action/citation_signature","submit_replication":"https://pith.science/pith/NYHYQYTDMNFPZCE3N6TKET2FS3/action/replication_record"}},"created_at":"2026-07-05T11:59:57.472024+00:00","updated_at":"2026-07-05T11:59:57.472024+00:00"}