{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BEDNU4QWOSU4QHIJ3CJIOTNR26","short_pith_number":"pith:BEDNU4QW","schema_version":"1.0","canonical_sha256":"0906da721674a9c81d09d892874db1d7baecfec2b48b499c8b4ec9230525919b","source":{"kind":"arxiv","id":"2408.06223","version":3},"attestation_state":"computed","paper":{"title":"On Effects of Steering Latent Representation for Large Language Model Unlearning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dang Huu-Tien, Hoang Thanh-Tung, Naoya Inoue, Trung-Tin Pham","submitted_at":"2024-08-12T15:24:50Z","abstract_excerpt":"Representation Misdirection for Unlearning (RMU), which steers model representation in the intermediate layer to a target random representation, is an effective method for large language model (LLM) unlearning. Despite its high performance, the underlying cause and explanation remain underexplored. In this paper, we theoretically demonstrate that steering forget representations in the intermediate layer reduces token confidence, causing LLMs to generate wrong or nonsense responses. We investigate how the coefficient influences the alignment of forget-sample representations with the random dire"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.06223","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-12T15:24:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d61e33701173637cef6d17d04ab713d877aa99de541ec3b8d3704ffaf5837ce6","abstract_canon_sha256":"a50b099b9d0ee22f9dd8689604827d4e2dea0ba3b9b2d4abe2f4307d3adbf483"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:08.489748Z","signature_b64":"CvVGpt4Dj+yLTmtwRRO/GOcsZOCuwG/KiQoG1TSjyYoPE2qMlPSUsxjgxgfk69RTcthgu+YhG0DxV7Hv5dFFAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0906da721674a9c81d09d892874db1d7baecfec2b48b499c8b4ec9230525919b","last_reissued_at":"2026-07-05T10:10:08.489252Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:08.489252Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Effects of Steering Latent Representation for Large Language Model Unlearning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dang Huu-Tien, Hoang Thanh-Tung, Naoya Inoue, Trung-Tin Pham","submitted_at":"2024-08-12T15:24:50Z","abstract_excerpt":"Representation Misdirection for Unlearning (RMU), which steers model representation in the intermediate layer to a target random representation, is an effective method for large language model (LLM) unlearning. Despite its high performance, the underlying cause and explanation remain underexplored. In this paper, we theoretically demonstrate that steering forget representations in the intermediate layer reduces token confidence, causing LLMs to generate wrong or nonsense responses. We investigate how the coefficient influences the alignment of forget-sample representations with the random dire"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.06223","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.06223/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.06223","created_at":"2026-07-05T10:10:08.489309+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.06223v3","created_at":"2026-07-05T10:10:08.489309+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.06223","created_at":"2026-07-05T10:10:08.489309+00:00"},{"alias_kind":"pith_short_12","alias_value":"BEDNU4QWOSU4","created_at":"2026-07-05T10:10:08.489309+00:00"},{"alias_kind":"pith_short_16","alias_value":"BEDNU4QWOSU4QHIJ","created_at":"2026-07-05T10:10:08.489309+00:00"},{"alias_kind":"pith_short_8","alias_value":"BEDNU4QW","created_at":"2026-07-05T10:10:08.489309+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12841","citing_title":"TimeROME-DLM: Temporal Causal Tracing and Low-Rank Inference-Time Knowledge Editing for Masked Diffusion Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12234","citing_title":"On The Effectiveness-Fluency Trade-Off In LLM Conditioning: A Systematic Study","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09559","citing_title":"Safe-RULE: Safe Reinforcement UnLEarning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26","json":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26.json","graph_json":"https://pith.science/api/pith-number/BEDNU4QWOSU4QHIJ3CJIOTNR26/graph.json","events_json":"https://pith.science/api/pith-number/BEDNU4QWOSU4QHIJ3CJIOTNR26/events.json","paper":"https://pith.science/paper/BEDNU4QW"},"agent_actions":{"view_html":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26","download_json":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26.json","view_paper":"https://pith.science/paper/BEDNU4QW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.06223&json=true","fetch_graph":"https://pith.science/api/pith-number/BEDNU4QWOSU4QHIJ3CJIOTNR26/graph.json","fetch_events":"https://pith.science/api/pith-number/BEDNU4QWOSU4QHIJ3CJIOTNR26/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26/action/storage_attestation","attest_author":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26/action/author_attestation","sign_citation":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26/action/citation_signature","submit_replication":"https://pith.science/pith/BEDNU4QWOSU4QHIJ3CJIOTNR26/action/replication_record"}},"created_at":"2026-07-05T10:10:08.489309+00:00","updated_at":"2026-07-05T10:10:08.489309+00:00"}