{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6G4TBLATTF4C7BPC32RT5QDWZK","short_pith_number":"pith:6G4TBLAT","schema_version":"1.0","canonical_sha256":"f1b930ac1399782f85e2dea33ec076caa1fe79b224101492dea062f31f9d8cfc","source":{"kind":"arxiv","id":"2505.15710","version":1},"attestation_state":"computed","paper":{"title":"Advancing LLM Safe Alignment with Safety Representation Ranking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenheng Zhang, Quan Chen, Tianqi Du, Yisen Wang, Zeming Wei","submitted_at":"2025-05-21T16:21:29Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has demonstrated milestone success in a variety of tasks, yet their potential for generating harmful content has raised significant safety concerns. Existing safety evaluation approaches typically operate directly on textual responses, overlooking the rich information embedded in the model's internal representations. In this paper, we propose Safety Representation Ranking (SRR), a listwise ranking framework that selects safe responses using hidden states from the LLM itself. SRR encodes both instructions and candidate completions using inte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15710","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T16:21:29Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6e6437b964d6eda29a8e68d037d724af42d1c395c0224064d099bfe6d02a4222","abstract_canon_sha256":"9f899459b155688a86f8ee2fe4b3c72b27f8e8899f0fa842f60fd8f631878910"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:49.872242Z","signature_b64":"ifBxVIDOCzLM8AWqf5e2s2y7n4Wk7rP74p5tYs0AlwCFUm34l6EbCYGiGWlrX5Mf7DnYZvGOm1qewAUtldCdDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1b930ac1399782f85e2dea33ec076caa1fe79b224101492dea062f31f9d8cfc","last_reissued_at":"2026-07-05T11:06:49.871617Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:49.871617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Advancing LLM Safe Alignment with Safety Representation Ranking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenheng Zhang, Quan Chen, Tianqi Du, Yisen Wang, Zeming Wei","submitted_at":"2025-05-21T16:21:29Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has demonstrated milestone success in a variety of tasks, yet their potential for generating harmful content has raised significant safety concerns. Existing safety evaluation approaches typically operate directly on textual responses, overlooking the rich information embedded in the model's internal representations. In this paper, we propose Safety Representation Ranking (SRR), a listwise ranking framework that selects safe responses using hidden states from the LLM itself. SRR encodes both instructions and candidate completions using inte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15710","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15710/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15710","created_at":"2026-07-05T11:06:49.871681+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15710v1","created_at":"2026-07-05T11:06:49.871681+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15710","created_at":"2026-07-05T11:06:49.871681+00:00"},{"alias_kind":"pith_short_12","alias_value":"6G4TBLATTF4C","created_at":"2026-07-05T11:06:49.871681+00:00"},{"alias_kind":"pith_short_16","alias_value":"6G4TBLATTF4C7BPC","created_at":"2026-07-05T11:06:49.871681+00:00"},{"alias_kind":"pith_short_8","alias_value":"6G4TBLAT","created_at":"2026-07-05T11:06:49.871681+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29629","citing_title":"Beyond Attack Success Rate: Temporal Logit Observability for LLM Safety Failures","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17504","citing_title":"A Distributional View for Visual Mechanistic Interpretability: KL-Minimal Soft-Constraint Principle","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01770","citing_title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02280","citing_title":"RACC: Representation-Aware Coverage Criteria for LLM Safety Testing","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11093","citing_title":"Enabling Performant and Flexible Model-Internal Observability for LLM Inference","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08881","citing_title":"Targeted Interpretable Safety Neuron Enhancement for Multilingual Vision-Language Large Models","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK","json":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK.json","graph_json":"https://pith.science/api/pith-number/6G4TBLATTF4C7BPC32RT5QDWZK/graph.json","events_json":"https://pith.science/api/pith-number/6G4TBLATTF4C7BPC32RT5QDWZK/events.json","paper":"https://pith.science/paper/6G4TBLAT"},"agent_actions":{"view_html":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK","download_json":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK.json","view_paper":"https://pith.science/paper/6G4TBLAT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15710&json=true","fetch_graph":"https://pith.science/api/pith-number/6G4TBLATTF4C7BPC32RT5QDWZK/graph.json","fetch_events":"https://pith.science/api/pith-number/6G4TBLATTF4C7BPC32RT5QDWZK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK/action/storage_attestation","attest_author":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK/action/author_attestation","sign_citation":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK/action/citation_signature","submit_replication":"https://pith.science/pith/6G4TBLATTF4C7BPC32RT5QDWZK/action/replication_record"}},"created_at":"2026-07-05T11:06:49.871681+00:00","updated_at":"2026-07-05T11:06:49.871681+00:00"}