{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4TLNOFZXGFH43H5VYMEK5GZYU6","short_pith_number":"pith:4TLNOFZX","schema_version":"1.0","canonical_sha256":"e4d6d71737314fcd9fb5c308ae9b38a7a39e6e79e517fa3cb2297e2739addb1f","source":{"kind":"arxiv","id":"2402.00402","version":1},"attestation_state":"computed","paper":{"title":"Investigating Bias Representations in Llama 2 Chat via Activation Steering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dawn Lu, Nina Rimsky","submitted_at":"2024-02-01T07:48:50Z","abstract_excerpt":"We address the challenge of societal bias in Large Language Models (LLMs), focusing on the Llama 2 7B Chat model. As LLMs are increasingly integrated into decision-making processes with substantial societal impact, it becomes imperative to ensure these models do not reinforce existing biases. Our approach employs activation steering to probe for and mitigate biases related to gender, race, and religion. This method manipulates model activations to direct responses towards or away from biased outputs, utilizing steering vectors derived from the StereoSet dataset and custom GPT4 generated gender"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00402","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-01T07:48:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b1b8d560492bc9a5d66c338e4fa64bbe068cbb92d4397745fcb5fc05947e391e","abstract_canon_sha256":"079dee84603e62617adb7b327f2f1c729f9ca9e6e809cfb386a25392e3fbff0a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:11.126379Z","signature_b64":"5KiEV6k0K8Y/VTe3KFj/2YRC4lmNCcYi3s7tU7fIVgFDbIuAVon5kRXd+I6eInNw6wB5UIC+Uti50LJ3O6GRCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4d6d71737314fcd9fb5c308ae9b38a7a39e6e79e517fa3cb2297e2739addb1f","last_reissued_at":"2026-07-05T07:40:11.125931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:11.125931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Investigating Bias Representations in Llama 2 Chat via Activation Steering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dawn Lu, Nina Rimsky","submitted_at":"2024-02-01T07:48:50Z","abstract_excerpt":"We address the challenge of societal bias in Large Language Models (LLMs), focusing on the Llama 2 7B Chat model. As LLMs are increasingly integrated into decision-making processes with substantial societal impact, it becomes imperative to ensure these models do not reinforce existing biases. Our approach employs activation steering to probe for and mitigate biases related to gender, race, and religion. This method manipulates model activations to direct responses towards or away from biased outputs, utilizing steering vectors derived from the StereoSet dataset and custom GPT4 generated gender"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00402","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00402/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00402","created_at":"2026-07-05T07:40:11.125987+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00402v1","created_at":"2026-07-05T07:40:11.125987+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00402","created_at":"2026-07-05T07:40:11.125987+00:00"},{"alias_kind":"pith_short_12","alias_value":"4TLNOFZXGFH4","created_at":"2026-07-05T07:40:11.125987+00:00"},{"alias_kind":"pith_short_16","alias_value":"4TLNOFZXGFH43H5V","created_at":"2026-07-05T07:40:11.125987+00:00"},{"alias_kind":"pith_short_8","alias_value":"4TLNOFZX","created_at":"2026-07-05T07:40:11.125987+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11599","citing_title":"When is Your LLM Steerable?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":198,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6","json":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6.json","graph_json":"https://pith.science/api/pith-number/4TLNOFZXGFH43H5VYMEK5GZYU6/graph.json","events_json":"https://pith.science/api/pith-number/4TLNOFZXGFH43H5VYMEK5GZYU6/events.json","paper":"https://pith.science/paper/4TLNOFZX"},"agent_actions":{"view_html":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6","download_json":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6.json","view_paper":"https://pith.science/paper/4TLNOFZX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00402&json=true","fetch_graph":"https://pith.science/api/pith-number/4TLNOFZXGFH43H5VYMEK5GZYU6/graph.json","fetch_events":"https://pith.science/api/pith-number/4TLNOFZXGFH43H5VYMEK5GZYU6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6/action/storage_attestation","attest_author":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6/action/author_attestation","sign_citation":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6/action/citation_signature","submit_replication":"https://pith.science/pith/4TLNOFZXGFH43H5VYMEK5GZYU6/action/replication_record"}},"created_at":"2026-07-05T07:40:11.125987+00:00","updated_at":"2026-07-05T07:40:11.125987+00:00"}