{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:N2LSZM44B47UY4FR4XC7JSHSJD","short_pith_number":"pith:N2LSZM44","schema_version":"1.0","canonical_sha256":"6e972cb39c0f3f4c70b1e5c5f4c8f248d8fa29bf820be48198f8ae179637430b","source":{"kind":"arxiv","id":"2504.02708","version":1},"attestation_state":"computed","paper":{"title":"The Hidden Space of Safety: Understanding Preference-Tuned LLMs in Multilingual context","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Manasa Bharadwaj, Nikhil Verma","submitted_at":"2025-04-03T15:46:46Z","abstract_excerpt":"Alignment tuning has enabled large language models to excel in reasoning, instruction-following, and minimizing harmful generations. However, despite their widespread deployment, these models exhibit a monolingual bias, raising concerns about the effectiveness of alignment across languages. Current alignment methods predominantly focus on English, leaving it unclear how alignment mechanism generalize to multilingual settings. To address this, we conduct a systematic analysis of distributional shifts in the embedding space of LLMs before and after alignment, uncovering its impact on model behav"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.02708","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-03T15:46:46Z","cross_cats_sorted":[],"title_canon_sha256":"2ea1933a9026bae2bf6189a4478152b8dc40f4aba237d29f3c4bb2057225d269","abstract_canon_sha256":"11f8680dfc45a43f12d727384f1ab42e0281e71d2ad8814f6e75c1e6193acbd1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:01.110158Z","signature_b64":"VdSpHFJ6z3eO+S9uesn1TLaaDYNwyk9kHid5s3hCZ8AazpTMT9AXck1CVJGuvr3sauBZtBndGI8MckSLp4tLBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e972cb39c0f3f4c70b1e5c5f4c8f248d8fa29bf820be48198f8ae179637430b","last_reissued_at":"2026-07-05T10:44:01.109685Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:01.109685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Hidden Space of Safety: Understanding Preference-Tuned LLMs in Multilingual context","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Manasa Bharadwaj, Nikhil Verma","submitted_at":"2025-04-03T15:46:46Z","abstract_excerpt":"Alignment tuning has enabled large language models to excel in reasoning, instruction-following, and minimizing harmful generations. However, despite their widespread deployment, these models exhibit a monolingual bias, raising concerns about the effectiveness of alignment across languages. Current alignment methods predominantly focus on English, leaving it unclear how alignment mechanism generalize to multilingual settings. To address this, we conduct a systematic analysis of distributional shifts in the embedding space of LLMs before and after alignment, uncovering its impact on model behav"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.02708","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.02708/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.02708","created_at":"2026-07-05T10:44:01.109744+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.02708v1","created_at":"2026-07-05T10:44:01.109744+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.02708","created_at":"2026-07-05T10:44:01.109744+00:00"},{"alias_kind":"pith_short_12","alias_value":"N2LSZM44B47U","created_at":"2026-07-05T10:44:01.109744+00:00"},{"alias_kind":"pith_short_16","alias_value":"N2LSZM44B47UY4FR","created_at":"2026-07-05T10:44:01.109744+00:00"},{"alias_kind":"pith_short_8","alias_value":"N2LSZM44","created_at":"2026-07-05T10:44:01.109744+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01196","citing_title":"Low-Resource Safety Failures Are Action Failures, Not Representation Failures","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD","json":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD.json","graph_json":"https://pith.science/api/pith-number/N2LSZM44B47UY4FR4XC7JSHSJD/graph.json","events_json":"https://pith.science/api/pith-number/N2LSZM44B47UY4FR4XC7JSHSJD/events.json","paper":"https://pith.science/paper/N2LSZM44"},"agent_actions":{"view_html":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD","download_json":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD.json","view_paper":"https://pith.science/paper/N2LSZM44","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.02708&json=true","fetch_graph":"https://pith.science/api/pith-number/N2LSZM44B47UY4FR4XC7JSHSJD/graph.json","fetch_events":"https://pith.science/api/pith-number/N2LSZM44B47UY4FR4XC7JSHSJD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD/action/storage_attestation","attest_author":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD/action/author_attestation","sign_citation":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD/action/citation_signature","submit_replication":"https://pith.science/pith/N2LSZM44B47UY4FR4XC7JSHSJD/action/replication_record"}},"created_at":"2026-07-05T10:44:01.109744+00:00","updated_at":"2026-07-05T10:44:01.109744+00:00"}