{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O4PHXD4RKXGIGLPJO777C5ESDC","short_pith_number":"pith:O4PHXD4R","schema_version":"1.0","canonical_sha256":"771e7b8f9155cc832de977fff174921887b44d6d37bf933d39e2c36e556710a5","source":{"kind":"arxiv","id":"2505.14352","version":1},"attestation_state":"computed","paper":{"title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bartosz Cywi\\'nski, Emil Ryd, Neel Nanda, Senthooran Rajamanoharan","submitted_at":"2025-05-20T13:36:37Z","abstract_excerpt":"As language models become more powerful and sophisticated, it is crucial that they remain trustworthy and reliable. There is concerning preliminary evidence that models may attempt to deceive or keep secrets from their operators. To explore the ability of current techniques to elicit such hidden knowledge, we train a Taboo model: a language model that describes a specific secret word without explicitly stating it. Importantly, the secret word is not presented to the model in its training data or prompt. We then investigate methods to uncover this secret. First, we evaluate non-interpretability"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14352","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-20T13:36:37Z","cross_cats_sorted":[],"title_canon_sha256":"df6bf043c7b1be4fc84cbdbba84258eb0a4cd52c383392fc2d5c03fad891fc72","abstract_canon_sha256":"29208a99399ca411cb6db332d2b96d9a6ee45733420e4a00d425913373a88054"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:00.093484Z","signature_b64":"qWZ+4sThLZPAsBnfavcI0sva3EUyMsBSTyk9eB1k948/4iKuNcB6qRRr9Pd69Sz7+q4YxyFnPudK8rXu/DNoBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"771e7b8f9155cc832de977fff174921887b44d6d37bf933d39e2c36e556710a5","last_reissued_at":"2026-07-05T11:06:00.093025Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:00.093025Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards eliciting latent knowledge from LLMs with mechanistic interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bartosz Cywi\\'nski, Emil Ryd, Neel Nanda, Senthooran Rajamanoharan","submitted_at":"2025-05-20T13:36:37Z","abstract_excerpt":"As language models become more powerful and sophisticated, it is crucial that they remain trustworthy and reliable. There is concerning preliminary evidence that models may attempt to deceive or keep secrets from their operators. To explore the ability of current techniques to elicit such hidden knowledge, we train a Taboo model: a language model that describes a specific secret word without explicitly stating it. Importantly, the secret word is not presented to the model in its training data or prompt. We then investigate methods to uncover this secret. First, we evaluate non-interpretability"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14352","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14352/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14352","created_at":"2026-07-05T11:06:00.093082+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14352v1","created_at":"2026-07-05T11:06:00.093082+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14352","created_at":"2026-07-05T11:06:00.093082+00:00"},{"alias_kind":"pith_short_12","alias_value":"O4PHXD4RKXGI","created_at":"2026-07-05T11:06:00.093082+00:00"},{"alias_kind":"pith_short_16","alias_value":"O4PHXD4RKXGIGLPJ","created_at":"2026-07-05T11:06:00.093082+00:00"},{"alias_kind":"pith_short_8","alias_value":"O4PHXD4R","created_at":"2026-07-05T11:06:00.093082+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18089","citing_title":"From Reasoning Traces to Reusable Modules: Understanding Compositional Generalization in Language Model Reasoning","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12618","citing_title":"\"Did you lie?\" Evaluating Lie Detectors across Model Scale and Belief-Verified Model Organisms","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01033","citing_title":"The Model Organism Lottery: Model Organism Interpretability Strongly Depends on Training Methodology","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00601","citing_title":"\"Don't Say It!\": Constraints, Compliance, and Communication when Language Models Play Taboo","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02609","citing_title":"Building Better Activation Oracles","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19270","citing_title":"DECOR: Auditing LLM Deception via Information Manipulation Theory","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC","json":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC.json","graph_json":"https://pith.science/api/pith-number/O4PHXD4RKXGIGLPJO777C5ESDC/graph.json","events_json":"https://pith.science/api/pith-number/O4PHXD4RKXGIGLPJO777C5ESDC/events.json","paper":"https://pith.science/paper/O4PHXD4R"},"agent_actions":{"view_html":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC","download_json":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC.json","view_paper":"https://pith.science/paper/O4PHXD4R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14352&json=true","fetch_graph":"https://pith.science/api/pith-number/O4PHXD4RKXGIGLPJO777C5ESDC/graph.json","fetch_events":"https://pith.science/api/pith-number/O4PHXD4RKXGIGLPJO777C5ESDC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC/action/storage_attestation","attest_author":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC/action/author_attestation","sign_citation":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC/action/citation_signature","submit_replication":"https://pith.science/pith/O4PHXD4RKXGIGLPJO777C5ESDC/action/replication_record"}},"created_at":"2026-07-05T11:06:00.093082+00:00","updated_at":"2026-07-05T11:06:00.093082+00:00"}