{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3CIDEMCT4R5VV63LUBBZNANF4T","short_pith_number":"pith:3CIDEMCT","schema_version":"1.0","canonical_sha256":"d890323053e47b5afb6ba0439681a5e4fea27f6c91e1ab1051d3390278f3bd04","source":{"kind":"arxiv","id":"2507.04673","version":1},"attestation_state":"computed","paper":{"title":"Trojan Horse Prompting: Jailbreaking Conversational Multimodal Models by Forging Assistant Message","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Li Qian, Wei Duan","submitted_at":"2025-07-07T05:35:21Z","abstract_excerpt":"The rise of conversational interfaces has greatly enhanced LLM usability by leveraging dialogue history for sophisticated reasoning. However, this reliance introduces an unexplored attack surface. This paper introduces Trojan Horse Prompting, a novel jailbreak technique. Adversaries bypass safety mechanisms by forging the model's own past utterances within the conversational history provided to its API. A malicious payload is injected into a model-attributed message, followed by a benign user prompt to trigger harmful content generation. This vulnerability stems from Asymmetric Safety Alignmen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.04673","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-07T05:35:21Z","cross_cats_sorted":[],"title_canon_sha256":"eb533bc45f152f8cb57172ab64c2f848de84b4d2ad6931c757a17d8382a8b876","abstract_canon_sha256":"81f846f8999f7c7d862a0b90c9253572f5d8eb630470f5785a62870bb60f8264"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:51.828685Z","signature_b64":"2lEHWOo5vOYlBzcFBVklHsBCGR1c8dFdCV0xT9tW4WLLFlrbA+4PhFwB4VSBjORIQsQWuiEajRQQzePDuTxmBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d890323053e47b5afb6ba0439681a5e4fea27f6c91e1ab1051d3390278f3bd04","last_reissued_at":"2026-07-05T11:32:51.828220Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:51.828220Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Trojan Horse Prompting: Jailbreaking Conversational Multimodal Models by Forging Assistant Message","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Li Qian, Wei Duan","submitted_at":"2025-07-07T05:35:21Z","abstract_excerpt":"The rise of conversational interfaces has greatly enhanced LLM usability by leveraging dialogue history for sophisticated reasoning. However, this reliance introduces an unexplored attack surface. This paper introduces Trojan Horse Prompting, a novel jailbreak technique. Adversaries bypass safety mechanisms by forging the model's own past utterances within the conversational history provided to its API. A malicious payload is injected into a model-attributed message, followed by a benign user prompt to trigger harmful content generation. This vulnerability stems from Asymmetric Safety Alignmen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.04673","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.04673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.04673","created_at":"2026-07-05T11:32:51.828274+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.04673v1","created_at":"2026-07-05T11:32:51.828274+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.04673","created_at":"2026-07-05T11:32:51.828274+00:00"},{"alias_kind":"pith_short_12","alias_value":"3CIDEMCT4R5V","created_at":"2026-07-05T11:32:51.828274+00:00"},{"alias_kind":"pith_short_16","alias_value":"3CIDEMCT4R5VV63L","created_at":"2026-07-05T11:32:51.828274+00:00"},{"alias_kind":"pith_short_8","alias_value":"3CIDEMCT","created_at":"2026-07-05T11:32:51.828274+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T","json":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T.json","graph_json":"https://pith.science/api/pith-number/3CIDEMCT4R5VV63LUBBZNANF4T/graph.json","events_json":"https://pith.science/api/pith-number/3CIDEMCT4R5VV63LUBBZNANF4T/events.json","paper":"https://pith.science/paper/3CIDEMCT"},"agent_actions":{"view_html":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T","download_json":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T.json","view_paper":"https://pith.science/paper/3CIDEMCT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.04673&json=true","fetch_graph":"https://pith.science/api/pith-number/3CIDEMCT4R5VV63LUBBZNANF4T/graph.json","fetch_events":"https://pith.science/api/pith-number/3CIDEMCT4R5VV63LUBBZNANF4T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T/action/storage_attestation","attest_author":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T/action/author_attestation","sign_citation":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T/action/citation_signature","submit_replication":"https://pith.science/pith/3CIDEMCT4R5VV63LUBBZNANF4T/action/replication_record"}},"created_at":"2026-07-05T11:32:51.828274+00:00","updated_at":"2026-07-05T11:32:51.828274+00:00"}