{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WK6UOGMU5O44YJ7O5Z6D75K5CE","short_pith_number":"pith:WK6UOGMU","schema_version":"1.0","canonical_sha256":"b2bd471994ebb9cc27eeee7c3ff55d111adf3f6e9a2164d83585c5e529adc67c","source":{"kind":"arxiv","id":"2503.09066","version":2},"attestation_state":"computed","paper":{"title":"Probing Latent Subspaces in LLM for AI Security: Identifying and Manipulating Adversarial States","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.LG","authors_text":"Jonathan Pan, Swee Liang Wong, Xin Wei Chia","submitted_at":"2025-03-12T04:59:22Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities across various tasks, yet they remain vulnerable to adversarial manipulations such as jailbreaking via prompt injection attacks. These attacks bypass safety mechanisms to generate restricted or harmful content. In this study, we investigated the underlying latent subspaces of safe and jailbroken states by extracting hidden activations from a LLM. Inspired by attractor dynamics in neuroscience, we hypothesized that LLM activations settle into semi stable states that can be identified and perturbed to induce state transitions"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.09066","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-12T04:59:22Z","cross_cats_sorted":["cs.AI","cs.CR"],"title_canon_sha256":"cd371ea0461b74864eb5779e684504a51119da5617f4715a75da28074a8bf416","abstract_canon_sha256":"52aaf6dabf6a4a824310ae8d0ac1a300fb488db93c084a51616931780b894d46"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:44.198343Z","signature_b64":"SWiCgASVxpnA9CHxue+Rm7JqoOohflRBvJax9nZqUlu1uRHMz3ITvspuWxTNIPUrxq7a6Kg8XKBa4Rj+FXI2Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b2bd471994ebb9cc27eeee7c3ff55d111adf3f6e9a2164d83585c5e529adc67c","last_reissued_at":"2026-07-05T11:31:44.197832Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:44.197832Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Probing Latent Subspaces in LLM for AI Security: Identifying and Manipulating Adversarial States","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR"],"primary_cat":"cs.LG","authors_text":"Jonathan Pan, Swee Liang Wong, Xin Wei Chia","submitted_at":"2025-03-12T04:59:22Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities across various tasks, yet they remain vulnerable to adversarial manipulations such as jailbreaking via prompt injection attacks. These attacks bypass safety mechanisms to generate restricted or harmful content. In this study, we investigated the underlying latent subspaces of safe and jailbroken states by extracting hidden activations from a LLM. Inspired by attractor dynamics in neuroscience, we hypothesized that LLM activations settle into semi stable states that can be identified and perturbed to induce state transitions"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.09066","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.09066/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.09066","created_at":"2026-07-05T11:31:44.197891+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.09066v2","created_at":"2026-07-05T11:31:44.197891+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.09066","created_at":"2026-07-05T11:31:44.197891+00:00"},{"alias_kind":"pith_short_12","alias_value":"WK6UOGMU5O44","created_at":"2026-07-05T11:31:44.197891+00:00"},{"alias_kind":"pith_short_16","alias_value":"WK6UOGMU5O44YJ7O","created_at":"2026-07-05T11:31:44.197891+00:00"},{"alias_kind":"pith_short_8","alias_value":"WK6UOGMU","created_at":"2026-07-05T11:31:44.197891+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27690","citing_title":"TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05741","citing_title":"HyperLens: Quantifying Cognitive Effort in LLMs with Fine-grained Confidence Trajectory","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19052","citing_title":"Cell-Based Representation of Relational Binding in Language Models","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE","json":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE.json","graph_json":"https://pith.science/api/pith-number/WK6UOGMU5O44YJ7O5Z6D75K5CE/graph.json","events_json":"https://pith.science/api/pith-number/WK6UOGMU5O44YJ7O5Z6D75K5CE/events.json","paper":"https://pith.science/paper/WK6UOGMU"},"agent_actions":{"view_html":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE","download_json":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE.json","view_paper":"https://pith.science/paper/WK6UOGMU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.09066&json=true","fetch_graph":"https://pith.science/api/pith-number/WK6UOGMU5O44YJ7O5Z6D75K5CE/graph.json","fetch_events":"https://pith.science/api/pith-number/WK6UOGMU5O44YJ7O5Z6D75K5CE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE/action/storage_attestation","attest_author":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE/action/author_attestation","sign_citation":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE/action/citation_signature","submit_replication":"https://pith.science/pith/WK6UOGMU5O44YJ7O5Z6D75K5CE/action/replication_record"}},"created_at":"2026-07-05T11:31:44.197891+00:00","updated_at":"2026-07-05T11:31:44.197891+00:00"}