{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6JVX2CUEW2SWNWSQIB2M6H2RSS","short_pith_number":"pith:6JVX2CUE","schema_version":"1.0","canonical_sha256":"f26b7d0a84b6a566da504074cf1f51949699a007f6062ed8ceb68298f083fdd0","source":{"kind":"arxiv","id":"2312.00374","version":3},"attestation_state":"computed","paper":{"title":"The Philosopher's Stone: Trojaning Plugins of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Guoxing Chen, Haojin Zhu, Minhui Xue, Rayne Holland, Shaofeng Li, Tian Dong, Yan Meng, Zhen Liu","submitted_at":"2023-12-01T06:36:17Z","abstract_excerpt":"Open-source Large Language Models (LLMs) have recently gained popularity because of their comparable performance to proprietary LLMs. To efficiently fulfill domain-specialized tasks, open-source LLMs can be refined, without expensive accelerators, using low-rank adapters. However, it is still unknown whether low-rank adapters can be exploited to control LLMs. To address this gap, we demonstrate that an infected adapter can induce, on specific triggers,an LLM to output content defined by an adversary and to even maliciously use tools. To train a Trojan adapter, we propose two novel attacks, POL"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.00374","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2023-12-01T06:36:17Z","cross_cats_sorted":[],"title_canon_sha256":"cdc2070424569801de3c578321f64dbf48da5cc3cc6492599eb1e8884e906a81","abstract_canon_sha256":"8f70a0908a4ab985456b416f2bc8375377bd8256a7263b66af4423391b71499a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:05:32.790503Z","signature_b64":"UklJobpkqG5RtxoPuE3ZEFkw4gHtlsPdkkkAMMPwuUx42jV+7zN5Qc3f6Zg5oytG36+SQ1hBAzgwbUaKUYInAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f26b7d0a84b6a566da504074cf1f51949699a007f6062ed8ceb68298f083fdd0","last_reissued_at":"2026-07-05T09:05:32.789986Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:05:32.789986Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Philosopher's Stone: Trojaning Plugins of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Guoxing Chen, Haojin Zhu, Minhui Xue, Rayne Holland, Shaofeng Li, Tian Dong, Yan Meng, Zhen Liu","submitted_at":"2023-12-01T06:36:17Z","abstract_excerpt":"Open-source Large Language Models (LLMs) have recently gained popularity because of their comparable performance to proprietary LLMs. To efficiently fulfill domain-specialized tasks, open-source LLMs can be refined, without expensive accelerators, using low-rank adapters. However, it is still unknown whether low-rank adapters can be exploited to control LLMs. To address this gap, we demonstrate that an infected adapter can induce, on specific triggers,an LLM to output content defined by an adversary and to even maliciously use tools. To train a Trojan adapter, we propose two novel attacks, POL"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.00374","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.00374/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.00374","created_at":"2026-07-05T09:05:32.790051+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.00374v3","created_at":"2026-07-05T09:05:32.790051+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.00374","created_at":"2026-07-05T09:05:32.790051+00:00"},{"alias_kind":"pith_short_12","alias_value":"6JVX2CUEW2SW","created_at":"2026-07-05T09:05:32.790051+00:00"},{"alias_kind":"pith_short_16","alias_value":"6JVX2CUEW2SWNWSQ","created_at":"2026-07-05T09:05:32.790051+00:00"},{"alias_kind":"pith_short_8","alias_value":"6JVX2CUE","created_at":"2026-07-05T09:05:32.790051+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.06556","citing_title":"Semantic Attacks on Tool-Augmented LLMs: Securing the Model Context Protocol Against Descriptor-Level Manipulation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02644","citing_title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05868","citing_title":"SkillScope: Toward Fine-Grained Least-Privilege Enforcement for Agent Skills","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS","json":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS.json","graph_json":"https://pith.science/api/pith-number/6JVX2CUEW2SWNWSQIB2M6H2RSS/graph.json","events_json":"https://pith.science/api/pith-number/6JVX2CUEW2SWNWSQIB2M6H2RSS/events.json","paper":"https://pith.science/paper/6JVX2CUE"},"agent_actions":{"view_html":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS","download_json":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS.json","view_paper":"https://pith.science/paper/6JVX2CUE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.00374&json=true","fetch_graph":"https://pith.science/api/pith-number/6JVX2CUEW2SWNWSQIB2M6H2RSS/graph.json","fetch_events":"https://pith.science/api/pith-number/6JVX2CUEW2SWNWSQIB2M6H2RSS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS/action/storage_attestation","attest_author":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS/action/author_attestation","sign_citation":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS/action/citation_signature","submit_replication":"https://pith.science/pith/6JVX2CUEW2SWNWSQIB2M6H2RSS/action/replication_record"}},"created_at":"2026-07-05T09:05:32.790051+00:00","updated_at":"2026-07-05T09:05:32.790051+00:00"}