{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WOJ6RXDR4ANB3EPQH6EW4MK7D7","short_pith_number":"pith:WOJ6RXDR","schema_version":"1.0","canonical_sha256":"b393e8dc71e01a1d91f03f896e315f1ffb54e017a70f69dbb4958642c92a5147","source":{"kind":"arxiv","id":"2406.07358","version":4},"attestation_state":"computed","paper":{"title":"AI Sandbagging: Language Models can Strategically Underperform on Evaluations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CY","cs.LG"],"primary_cat":"cs.AI","authors_text":"Felix Hofst\\\"atter, Francis Rhys Ward, Ollie Jaffe, Samuel F. Brown, Teun van der Weij","submitted_at":"2024-06-11T15:26:57Z","abstract_excerpt":"Trustworthy capability evaluations are crucial for ensuring the safety of AI systems, and are becoming a key component of AI regulation. However, the developers of an AI system, or the AI system itself, may have incentives for evaluations to understate the AI's actual capability. These conflicting interests lead to the problem of sandbagging, which we define as strategic underperformance on an evaluation. In this paper we assess sandbagging capabilities in contemporary language models (LMs). We prompt frontier LMs, like GPT-4 and Claude 3 Opus, to selectively underperform on dangerous capabili"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07358","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-11T15:26:57Z","cross_cats_sorted":["cs.CL","cs.CY","cs.LG"],"title_canon_sha256":"acd608ef5917620af19fbcc26f2d7d7fba7a8cbfd1bb0ff98944008b74f9cea8","abstract_canon_sha256":"35a2d5f19b0e3362958db79a4c43b729153a18427c77f3c8f8ef48c08c3f2337"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:41.147047Z","signature_b64":"JeZpW6QEdwNxEDmU71APVIrUGRTpjZifjxV3DzKSo0fk4xuwiejY0kfseSROUWO+kKL9ZiR9o61PtUSeNmdQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b393e8dc71e01a1d91f03f896e315f1ffb54e017a70f69dbb4958642c92a5147","last_reissued_at":"2026-07-05T10:10:41.146617Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:41.146617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AI Sandbagging: Language Models can Strategically Underperform on Evaluations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CY","cs.LG"],"primary_cat":"cs.AI","authors_text":"Felix Hofst\\\"atter, Francis Rhys Ward, Ollie Jaffe, Samuel F. Brown, Teun van der Weij","submitted_at":"2024-06-11T15:26:57Z","abstract_excerpt":"Trustworthy capability evaluations are crucial for ensuring the safety of AI systems, and are becoming a key component of AI regulation. However, the developers of an AI system, or the AI system itself, may have incentives for evaluations to understate the AI's actual capability. These conflicting interests lead to the problem of sandbagging, which we define as strategic underperformance on an evaluation. In this paper we assess sandbagging capabilities in contemporary language models (LMs). We prompt frontier LMs, like GPT-4 and Claude 3 Opus, to selectively underperform on dangerous capabili"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07358","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07358","created_at":"2026-07-05T10:10:41.146670+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07358v4","created_at":"2026-07-05T10:10:41.146670+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07358","created_at":"2026-07-05T10:10:41.146670+00:00"},{"alias_kind":"pith_short_12","alias_value":"WOJ6RXDR4ANB","created_at":"2026-07-05T10:10:41.146670+00:00"},{"alias_kind":"pith_short_16","alias_value":"WOJ6RXDR4ANB3EPQ","created_at":"2026-07-05T10:10:41.146670+00:00"},{"alias_kind":"pith_short_8","alias_value":"WOJ6RXDR","created_at":"2026-07-05T10:10:41.146670+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08066","citing_title":"Persuasion Attacks Can Decrease Effectiveness of CoT Monitoring","ref_index":111,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01033","citing_title":"The Model Organism Lottery: Model Organism Interpretability Strongly Depends on Training Methodology","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06529","citing_title":"Attack Selection in Agentic AI Control Evaluations Meaningfully Decreases Safety","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02211","citing_title":"Consistency Training while Mitigating Obfuscation via Rate Matching","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07631","citing_title":"Trait-space Monitoring for Emergent Misalignment During Supervised Finetuning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30916","citing_title":"Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06286","citing_title":"LLMs Can Leak Training Data But Do They Want To? A Propensity-Aware Evaluation of Memorization in LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17408","citing_title":"The Impact of Off-Policy Training Data on Probe Generalisation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01166","citing_title":"Evaluating AI Providers' Frontier Safety Frameworks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04984","citing_title":"Frontier Models are Capable of In-context Scheming","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13329","citing_title":"Tracing Persona Vectors Through LLM Pretraining","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26206","citing_title":"Option-Order Randomisation Reveals a Distributional Position Attractor in Prompted Sandbagging","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10601","citing_title":"The Open-Box Fallacy: Why AI Deployment Needs a Calibrated Verification Regime","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25249","citing_title":"Below-Chance Blindness: Prompted Underperformance in Small LLMs Produces Positional Bias Rather than Answer Avoidance","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24966","citing_title":"Risk Reporting for Developers' Internal AI Model Use","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05835","citing_title":"Evaluation Awareness in Language Models Has Limited Effect on Behaviour","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09104","citing_title":"Scheming in the wild: detecting real-world AI scheming incidents with open-source intelligence","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17663","citing_title":"ATLAS: Constitution-Conditioned Latent Geometry and Redistribution Across Language Models and Neural Perturbation Data","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06327","citing_title":"Measuring Evaluation-Context Divergence in Open-Weight LLMs: A Paired-Prompt Protocol with Pilot Evidence of Alignment-Pipeline-Specific Heterogeneity","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7","json":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7.json","graph_json":"https://pith.science/api/pith-number/WOJ6RXDR4ANB3EPQH6EW4MK7D7/graph.json","events_json":"https://pith.science/api/pith-number/WOJ6RXDR4ANB3EPQH6EW4MK7D7/events.json","paper":"https://pith.science/paper/WOJ6RXDR"},"agent_actions":{"view_html":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7","download_json":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7.json","view_paper":"https://pith.science/paper/WOJ6RXDR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07358&json=true","fetch_graph":"https://pith.science/api/pith-number/WOJ6RXDR4ANB3EPQH6EW4MK7D7/graph.json","fetch_events":"https://pith.science/api/pith-number/WOJ6RXDR4ANB3EPQH6EW4MK7D7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7/action/storage_attestation","attest_author":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7/action/author_attestation","sign_citation":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7/action/citation_signature","submit_replication":"https://pith.science/pith/WOJ6RXDR4ANB3EPQH6EW4MK7D7/action/replication_record"}},"created_at":"2026-07-05T10:10:41.146670+00:00","updated_at":"2026-07-05T10:10:41.146670+00:00"}