{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IB7C2R7VWDSN54OONL72B4MUY4","short_pith_number":"pith:IB7C2R7V","schema_version":"1.0","canonical_sha256":"407e2d47f5b0e4def1ce6affa0f194c707ab0cd58c9dae21e84d59601cb77ede","source":{"kind":"arxiv","id":"2311.07587","version":2},"attestation_state":"computed","paper":{"title":"Frontier Language Models are not Robust to Adversarial Arithmetic, or \"What do I need to say so you agree 2+2=5?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Parisi, Alex Alemi, Alex Rizkowsky, Azade Nova, Ben Adlam, Bernd Bohnet, C. Daniel Freeman, Gamaleldin F Elsayed, Gaurav Mishra, Hanie Sedghi, Igor Mordatch, Isabelle Simpson, Izzeddin Gur, Jaehoon Lee, Jascha Sohl-Dickstein, JD Co-Reyes, Jeffrey Pennington, Kelvin Xu, Kevin Swersky, Kshiteej Mahajan, Laura Culp, Lechao Xiao, Maxwell L Bileschi, Noah Constant, Noah Fiedel, Peter J. Liu, Roman Novak, Rosanne Liu, Simon Kornblith, Yundi Qian","submitted_at":"2023-11-08T19:07:10Z","abstract_excerpt":"We introduce and study the problem of adversarial arithmetic, which provides a simple yet challenging testbed for language model alignment. This problem is comprised of arithmetic questions posed in natural language, with an arbitrary adversarial string inserted before the question is complete. Even in the simple setting of 1-digit addition problems, it is easy to find adversarial prompts that make all tested models (including PaLM2, GPT4, Claude2) misbehave, and even to steer models to a particular wrong answer. We additionally provide a simple algorithm for finding successful attacks by quer"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07587","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-08T19:07:10Z","cross_cats_sorted":["cs.AI","cs.CY","cs.LG"],"title_canon_sha256":"d0dfa87da6e9159d0c93fd11d2762fb02488d05989bbfa1adb43fd9d5d6aa7c8","abstract_canon_sha256":"056a49b94184e713edf8cdd9b496f5055da051685f033490e1d28e0f09555646"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:13:11.309762Z","signature_b64":"vZVU6YZJWn5/lcXhBJa+/dXw69brLYZd5i7rwmpBE3HLiRJwVySK3HDbjpnIlcFNZBw/MrDUyWYMdFqc11WEBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"407e2d47f5b0e4def1ce6affa0f194c707ab0cd58c9dae21e84d59601cb77ede","last_reissued_at":"2026-07-05T07:13:11.309212Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:13:11.309212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Frontier Language Models are not Robust to Adversarial Arithmetic, or \"What do I need to say so you agree 2+2=5?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Parisi, Alex Alemi, Alex Rizkowsky, Azade Nova, Ben Adlam, Bernd Bohnet, C. Daniel Freeman, Gamaleldin F Elsayed, Gaurav Mishra, Hanie Sedghi, Igor Mordatch, Isabelle Simpson, Izzeddin Gur, Jaehoon Lee, Jascha Sohl-Dickstein, JD Co-Reyes, Jeffrey Pennington, Kelvin Xu, Kevin Swersky, Kshiteej Mahajan, Laura Culp, Lechao Xiao, Maxwell L Bileschi, Noah Constant, Noah Fiedel, Peter J. Liu, Roman Novak, Rosanne Liu, Simon Kornblith, Yundi Qian","submitted_at":"2023-11-08T19:07:10Z","abstract_excerpt":"We introduce and study the problem of adversarial arithmetic, which provides a simple yet challenging testbed for language model alignment. This problem is comprised of arithmetic questions posed in natural language, with an arbitrary adversarial string inserted before the question is complete. Even in the simple setting of 1-digit addition problems, it is easy to find adversarial prompts that make all tested models (including PaLM2, GPT4, Claude2) misbehave, and even to steer models to a particular wrong answer. We additionally provide a simple algorithm for finding successful attacks by quer"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07587","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07587/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07587","created_at":"2026-07-05T07:13:11.309277+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07587v2","created_at":"2026-07-05T07:13:11.309277+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07587","created_at":"2026-07-05T07:13:11.309277+00:00"},{"alias_kind":"pith_short_12","alias_value":"IB7C2R7VWDSN","created_at":"2026-07-05T07:13:11.309277+00:00"},{"alias_kind":"pith_short_16","alias_value":"IB7C2R7VWDSN54OO","created_at":"2026-07-05T07:13:11.309277+00:00"},{"alias_kind":"pith_short_8","alias_value":"IB7C2R7V","created_at":"2026-07-05T07:13:11.309277+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08068","citing_title":"DICE: Entropy-Regularized Equilibrium Selection for Stable Multi-Agent LLM Coordination","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4","json":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4.json","graph_json":"https://pith.science/api/pith-number/IB7C2R7VWDSN54OONL72B4MUY4/graph.json","events_json":"https://pith.science/api/pith-number/IB7C2R7VWDSN54OONL72B4MUY4/events.json","paper":"https://pith.science/paper/IB7C2R7V"},"agent_actions":{"view_html":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4","download_json":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4.json","view_paper":"https://pith.science/paper/IB7C2R7V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07587&json=true","fetch_graph":"https://pith.science/api/pith-number/IB7C2R7VWDSN54OONL72B4MUY4/graph.json","fetch_events":"https://pith.science/api/pith-number/IB7C2R7VWDSN54OONL72B4MUY4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4/action/storage_attestation","attest_author":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4/action/author_attestation","sign_citation":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4/action/citation_signature","submit_replication":"https://pith.science/pith/IB7C2R7VWDSN54OONL72B4MUY4/action/replication_record"}},"created_at":"2026-07-05T07:13:11.309277+00:00","updated_at":"2026-07-05T07:13:11.309277+00:00"}