{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:AB77H72RFCBVDLAVWHBEGDHWGK","short_pith_number":"pith:AB77H72R","schema_version":"1.0","canonical_sha256":"007ff3ff51288351ac15b1c2430cf632b2598cfc71399fa693720d2a7ff9ce70","source":{"kind":"arxiv","id":"2607.17890","version":1},"attestation_state":"computed","paper":{"title":"Stress Testing Concept Erasure with Large Language Model Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Edward Moroshko, Feng Chen, Jingyu Sun, Sotirios A. Tsaftaris, Steven McDonagh, Yuyang Xue, Zhihua Liu","submitted_at":"2026-07-20T12:38:50Z","abstract_excerpt":"Concept erasure aims to remove semantic concepts from a trained generative model and is increasingly important for responsible AI deployment. However, verifying whether a model has robustly removed targeted concepts remains a critical challenge. Existing evaluation methods are typically pre-defined and static, failing to expose vulnerabilities under diverse natural-language probes and challenging conditions. Moreover, manually designed evaluation strategies can be biased and difficult to scale. We posit that concept erasure evaluation is best formulated as an adaptive hypothesis search, operat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.17890","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-20T12:38:50Z","cross_cats_sorted":[],"title_canon_sha256":"39918c2512fa9dfad8118992b670cb2af0463972b2638b594200f7fc953e3e84","abstract_canon_sha256":"93ac979fd6306d1b909695c21d4b8bfff49ea6f14eafe72f483a7d8af0aa5acd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T02:22:05.240734Z","signature_b64":"X7guGqYPVAkANkAFrHCZMVFi9RVtmW1OzziZNV3PO/47JgTjTSEkd1XkZQzHnr4jKk0X3Oru6IOqfEx6VVgvDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"007ff3ff51288351ac15b1c2430cf632b2598cfc71399fa693720d2a7ff9ce70","last_reissued_at":"2026-07-21T02:22:05.239898Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T02:22:05.239898Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Stress Testing Concept Erasure with Large Language Model Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Edward Moroshko, Feng Chen, Jingyu Sun, Sotirios A. Tsaftaris, Steven McDonagh, Yuyang Xue, Zhihua Liu","submitted_at":"2026-07-20T12:38:50Z","abstract_excerpt":"Concept erasure aims to remove semantic concepts from a trained generative model and is increasingly important for responsible AI deployment. However, verifying whether a model has robustly removed targeted concepts remains a critical challenge. Existing evaluation methods are typically pre-defined and static, failing to expose vulnerabilities under diverse natural-language probes and challenging conditions. Moreover, manually designed evaluation strategies can be biased and difficult to scale. We posit that concept erasure evaluation is best formulated as an adaptive hypothesis search, operat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.17890","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.17890/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.17890","created_at":"2026-07-21T02:22:05.240336+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.17890v1","created_at":"2026-07-21T02:22:05.240336+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.17890","created_at":"2026-07-21T02:22:05.240336+00:00"},{"alias_kind":"pith_short_12","alias_value":"AB77H72RFCBV","created_at":"2026-07-21T02:22:05.240336+00:00"},{"alias_kind":"pith_short_16","alias_value":"AB77H72RFCBVDLAV","created_at":"2026-07-21T02:22:05.240336+00:00"},{"alias_kind":"pith_short_8","alias_value":"AB77H72R","created_at":"2026-07-21T02:22:05.240336+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK","json":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK.json","graph_json":"https://pith.science/api/pith-number/AB77H72RFCBVDLAVWHBEGDHWGK/graph.json","events_json":"https://pith.science/api/pith-number/AB77H72RFCBVDLAVWHBEGDHWGK/events.json","paper":"https://pith.science/paper/AB77H72R"},"agent_actions":{"view_html":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK","download_json":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK.json","view_paper":"https://pith.science/paper/AB77H72R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.17890&json=true","fetch_graph":"https://pith.science/api/pith-number/AB77H72RFCBVDLAVWHBEGDHWGK/graph.json","fetch_events":"https://pith.science/api/pith-number/AB77H72RFCBVDLAVWHBEGDHWGK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK/action/storage_attestation","attest_author":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK/action/author_attestation","sign_citation":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK/action/citation_signature","submit_replication":"https://pith.science/pith/AB77H72RFCBVDLAVWHBEGDHWGK/action/replication_record"}},"created_at":"2026-07-21T02:22:05.240336+00:00","updated_at":"2026-07-21T02:22:05.240336+00:00"}