{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZFI6AH67QUPXMDGP5MYJQEXDG7","short_pith_number":"pith:ZFI6AH67","schema_version":"1.0","canonical_sha256":"c951e01fdf851f760ccfeb309812e337c73ab93f4b5a68b3f10104f8fd0de1fb","source":{"kind":"arxiv","id":"2502.12025","version":1},"attestation_state":"computed","paper":{"title":"SafeChain: Safety of Language Models with Long Chain-of-Thought Reasoning Capabilities","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Bill Yuchen Lin, Bo Li, Fengqing Jiang, Luyao Niu, Radha Poovendran, Yuetai Li, Zhangchen Xu, Zhen Xiang","submitted_at":"2025-02-17T16:57:56Z","abstract_excerpt":"Emerging large reasoning models (LRMs), such as DeepSeek-R1 models, leverage long chain-of-thought (CoT) reasoning to generate structured intermediate steps, enhancing their reasoning capabilities. However, long CoT does not inherently guarantee safe outputs, potentially leading to harmful consequences such as the introduction of security vulnerabilities in code or the spread of misinformation. Current research on large language model (LLM) safety usually focuses on short-answer responses, overlooking the long CoT style outputs of LRMs. To bridge this gap, we conduct a systematic study of LRM "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.12025","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-17T16:57:56Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a14724af253a7304a6dca4b59d334401366ac39870e23f00e873451a562d2b4c","abstract_canon_sha256":"96ef75813bb00abd150cc44a7f8803783c7ef3cfd3046b003e0262cf32f00dde"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:37.540444Z","signature_b64":"376oI1M4XpNbw/MVZCFC9bhunq6FBmQq7xJkh950CHIi5fyaDti0XcFtq6bzhN0ZzGuxtfMo3UhBHbt0+flWAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c951e01fdf851f760ccfeb309812e337c73ab93f4b5a68b3f10104f8fd0de1fb","last_reissued_at":"2026-07-05T10:15:37.539870Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:37.539870Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SafeChain: Safety of Language Models with Long Chain-of-Thought Reasoning Capabilities","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Bill Yuchen Lin, Bo Li, Fengqing Jiang, Luyao Niu, Radha Poovendran, Yuetai Li, Zhangchen Xu, Zhen Xiang","submitted_at":"2025-02-17T16:57:56Z","abstract_excerpt":"Emerging large reasoning models (LRMs), such as DeepSeek-R1 models, leverage long chain-of-thought (CoT) reasoning to generate structured intermediate steps, enhancing their reasoning capabilities. However, long CoT does not inherently guarantee safe outputs, potentially leading to harmful consequences such as the introduction of security vulnerabilities in code or the spread of misinformation. Current research on large language model (LLM) safety usually focuses on short-answer responses, overlooking the long CoT style outputs of LRMs. To bridge this gap, we conduct a systematic study of LRM "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.12025","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.12025/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.12025","created_at":"2026-07-05T10:15:37.539942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.12025v1","created_at":"2026-07-05T10:15:37.539942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.12025","created_at":"2026-07-05T10:15:37.539942+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZFI6AH67QUPX","created_at":"2026-07-05T10:15:37.539942+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZFI6AH67QUPXMDGP","created_at":"2026-07-05T10:15:37.539942+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZFI6AH67","created_at":"2026-07-05T10:15:37.539942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31748","citing_title":"Addressing Over-Refusal in LLMs with Competing Rewards","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04204","citing_title":"ReasoningGuard: Safeguarding Large Reasoning Models with Inference-time Safety Aha Moments","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21882","citing_title":"Position: The Hidden Costs and Measurement Gaps of Reinforcement Learning with Verifiable Rewards","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21285","citing_title":"When Models Outthink Their Safety: Unveiling and Mitigating Self-Jailbreak in Large Reasoning Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2504.21318","citing_title":"Phi-4-reasoning Technical Report","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03332","citing_title":"Fragile Thoughts: How Large Language Models Handle Chain-of-Thought Perturbations","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01687","citing_title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18946","citing_title":"Reasoning Structure Matters for Safety Alignment of Reasoning Models","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7","json":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7.json","graph_json":"https://pith.science/api/pith-number/ZFI6AH67QUPXMDGP5MYJQEXDG7/graph.json","events_json":"https://pith.science/api/pith-number/ZFI6AH67QUPXMDGP5MYJQEXDG7/events.json","paper":"https://pith.science/paper/ZFI6AH67"},"agent_actions":{"view_html":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7","download_json":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7.json","view_paper":"https://pith.science/paper/ZFI6AH67","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.12025&json=true","fetch_graph":"https://pith.science/api/pith-number/ZFI6AH67QUPXMDGP5MYJQEXDG7/graph.json","fetch_events":"https://pith.science/api/pith-number/ZFI6AH67QUPXMDGP5MYJQEXDG7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7/action/storage_attestation","attest_author":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7/action/author_attestation","sign_citation":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7/action/citation_signature","submit_replication":"https://pith.science/pith/ZFI6AH67QUPXMDGP5MYJQEXDG7/action/replication_record"}},"created_at":"2026-07-05T10:15:37.539942+00:00","updated_at":"2026-07-05T10:15:37.539942+00:00"}