{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:V2AGDCTIFYTKWKIUUXUBVXRMI7","short_pith_number":"pith:V2AGDCTI","schema_version":"1.0","canonical_sha256":"ae80618a682e26ab2914a5e81ade2c47c6097451c44f9037f8caf8a2be02b594","source":{"kind":"arxiv","id":"2502.01236","version":1},"attestation_state":"computed","paper":{"title":"Eliciting Language Model Behaviors with Investigator Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Daniel D. Johnson, Jacob Steinhardt, Neil Chowdhury, Percy Liang, Sarah Schwettmann, Tatsunori Hashimoto, Xiang Lisa Li","submitted_at":"2025-02-03T10:52:44Z","abstract_excerpt":"Language models exhibit complex, diverse behaviors when prompted with free-form text, making it difficult to characterize the space of possible outputs. We study the problem of behavior elicitation, where the goal is to search for prompts that induce specific target behaviors (e.g., hallucinations or harmful responses) from a target language model. To navigate the exponentially large space of possible prompts, we train investigator models to map randomly-chosen target behaviors to a diverse distribution of outputs that elicit them, similar to amortized Bayesian inference. We do this through su"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01236","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T10:52:44Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"37f936c4a86df33e22627c5dffdeeabc00582801f918b109e94adaa6be785986","abstract_canon_sha256":"f582288e9a15696eb6667193c4b4b1baa86cd829321b02a63e26caf843a36e5c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:44.651849Z","signature_b64":"nZGtdWuCIuJgY/Vbgua2gQBaoBxzYE0xbApid5KCJ2+7zBckD+cjoot4H/nUhdqjBBwyCsHN3/p5y3OHlKDSAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae80618a682e26ab2914a5e81ade2c47c6097451c44f9037f8caf8a2be02b594","last_reissued_at":"2026-07-05T10:08:44.651403Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:44.651403Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Eliciting Language Model Behaviors with Investigator Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Daniel D. Johnson, Jacob Steinhardt, Neil Chowdhury, Percy Liang, Sarah Schwettmann, Tatsunori Hashimoto, Xiang Lisa Li","submitted_at":"2025-02-03T10:52:44Z","abstract_excerpt":"Language models exhibit complex, diverse behaviors when prompted with free-form text, making it difficult to characterize the space of possible outputs. We study the problem of behavior elicitation, where the goal is to search for prompts that induce specific target behaviors (e.g., hallucinations or harmful responses) from a target language model. To navigate the exponentially large space of possible prompts, we train investigator models to map randomly-chosen target behaviors to a diverse distribution of outputs that elicit them, similar to amortized Bayesian inference. We do this through su"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01236","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01236/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01236","created_at":"2026-07-05T10:08:44.651461+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01236v1","created_at":"2026-07-05T10:08:44.651461+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01236","created_at":"2026-07-05T10:08:44.651461+00:00"},{"alias_kind":"pith_short_12","alias_value":"V2AGDCTIFYTK","created_at":"2026-07-05T10:08:44.651461+00:00"},{"alias_kind":"pith_short_16","alias_value":"V2AGDCTIFYTKWKIU","created_at":"2026-07-05T10:08:44.651461+00:00"},{"alias_kind":"pith_short_8","alias_value":"V2AGDCTI","created_at":"2026-07-05T10:08:44.651461+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12299","citing_title":"Learning What to Say to Your VLA: Mostly Harmless Vision Language Action Model Steering","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":141,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7","json":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7.json","graph_json":"https://pith.science/api/pith-number/V2AGDCTIFYTKWKIUUXUBVXRMI7/graph.json","events_json":"https://pith.science/api/pith-number/V2AGDCTIFYTKWKIUUXUBVXRMI7/events.json","paper":"https://pith.science/paper/V2AGDCTI"},"agent_actions":{"view_html":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7","download_json":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7.json","view_paper":"https://pith.science/paper/V2AGDCTI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01236&json=true","fetch_graph":"https://pith.science/api/pith-number/V2AGDCTIFYTKWKIUUXUBVXRMI7/graph.json","fetch_events":"https://pith.science/api/pith-number/V2AGDCTIFYTKWKIUUXUBVXRMI7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7/action/storage_attestation","attest_author":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7/action/author_attestation","sign_citation":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7/action/citation_signature","submit_replication":"https://pith.science/pith/V2AGDCTIFYTKWKIUUXUBVXRMI7/action/replication_record"}},"created_at":"2026-07-05T10:08:44.651461+00:00","updated_at":"2026-07-05T10:08:44.651461+00:00"}