{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BBADCAGQ75XMG6WUSZDYSXSFCJ","short_pith_number":"pith:BBADCAGQ","schema_version":"1.0","canonical_sha256":"08403100d0ff6ec37ad49647895e451258ad94289c57ead30343b71440f44aa7","source":{"kind":"arxiv","id":"2403.14988","version":1},"attestation_state":"computed","paper":{"title":"Risk and Response in Large Language Models: Evaluating Key Threat Categories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abel Salinas, Bahareh Harandizadeh, Fred Morstatter","submitted_at":"2024-03-22T06:46:40Z","abstract_excerpt":"This paper explores the pressing issue of risk assessment in Large Language Models (LLMs) as they become increasingly prevalent in various applications. Focusing on how reward models, which are designed to fine-tune pretrained LLMs to align with human values, perceive and categorize different types of risks, we delve into the challenges posed by the subjective nature of preference-based training data. By utilizing the Anthropic Red-team dataset, we analyze major risk categories, including Information Hazards, Malicious Uses, and Discrimination/Hateful content. Our findings indicate that LLMs t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14988","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-22T06:46:40Z","cross_cats_sorted":[],"title_canon_sha256":"5059653aa25bb473eef3f118a6973566209dd69d3bb43c6ca43b4bc655db9258","abstract_canon_sha256":"b577b79814bd412ae6cf0bf393bc77ec1e663650123db88144bf4ec376ea24fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:28.827532Z","signature_b64":"jwyURiHmAQbW/P4Ej0w7JoA2+8T8rxn0FjvSXTjVLF37ETLEJK8CCqNdPNNT4zqyBABU7w2GcyeiGA3nvYmiBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"08403100d0ff6ec37ad49647895e451258ad94289c57ead30343b71440f44aa7","last_reissued_at":"2026-07-05T07:59:28.827124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:28.827124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Risk and Response in Large Language Models: Evaluating Key Threat Categories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abel Salinas, Bahareh Harandizadeh, Fred Morstatter","submitted_at":"2024-03-22T06:46:40Z","abstract_excerpt":"This paper explores the pressing issue of risk assessment in Large Language Models (LLMs) as they become increasingly prevalent in various applications. Focusing on how reward models, which are designed to fine-tune pretrained LLMs to align with human values, perceive and categorize different types of risks, we delve into the challenges posed by the subjective nature of preference-based training data. By utilizing the Anthropic Red-team dataset, we analyze major risk categories, including Information Hazards, Malicious Uses, and Discrimination/Hateful content. Our findings indicate that LLMs t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14988","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14988/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14988","created_at":"2026-07-05T07:59:28.827184+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14988v1","created_at":"2026-07-05T07:59:28.827184+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14988","created_at":"2026-07-05T07:59:28.827184+00:00"},{"alias_kind":"pith_short_12","alias_value":"BBADCAGQ75XM","created_at":"2026-07-05T07:59:28.827184+00:00"},{"alias_kind":"pith_short_16","alias_value":"BBADCAGQ75XMG6WU","created_at":"2026-07-05T07:59:28.827184+00:00"},{"alias_kind":"pith_short_8","alias_value":"BBADCAGQ","created_at":"2026-07-05T07:59:28.827184+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.19165","citing_title":"OrgAccess: A Benchmark for Role Based Access Control in Organization Scale LLMs","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ","json":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ.json","graph_json":"https://pith.science/api/pith-number/BBADCAGQ75XMG6WUSZDYSXSFCJ/graph.json","events_json":"https://pith.science/api/pith-number/BBADCAGQ75XMG6WUSZDYSXSFCJ/events.json","paper":"https://pith.science/paper/BBADCAGQ"},"agent_actions":{"view_html":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ","download_json":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ.json","view_paper":"https://pith.science/paper/BBADCAGQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14988&json=true","fetch_graph":"https://pith.science/api/pith-number/BBADCAGQ75XMG6WUSZDYSXSFCJ/graph.json","fetch_events":"https://pith.science/api/pith-number/BBADCAGQ75XMG6WUSZDYSXSFCJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ/action/storage_attestation","attest_author":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ/action/author_attestation","sign_citation":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ/action/citation_signature","submit_replication":"https://pith.science/pith/BBADCAGQ75XMG6WUSZDYSXSFCJ/action/replication_record"}},"created_at":"2026-07-05T07:59:28.827184+00:00","updated_at":"2026-07-05T07:59:28.827184+00:00"}