{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UEHYY3HMI2UZEDPOQ5HMGMLYUS","short_pith_number":"pith:UEHYY3HM","schema_version":"1.0","canonical_sha256":"a10f8c6cec46a9920dee874ec33178a49a4ec62a57210ce25c2a5bec206c9637","source":{"kind":"arxiv","id":"2505.10937","version":1},"attestation_state":"computed","paper":{"title":"Reasoning with OmniThought: A Large CoT Dataset with Verbosity and Cognitive Difficulty Annotations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chengyu Wang, Junbing Yan, Jun Huang, Wenrui Cai, Xiangzhong Fang","submitted_at":"2025-05-16T07:15:30Z","abstract_excerpt":"The emergence of large reasoning models (LRMs) has transformed Natural Language Processing by excelling in complex tasks such as mathematical problem-solving and code generation. These models leverage chain-of-thought (CoT) processes, enabling them to emulate human-like reasoning strategies. However, the advancement of LRMs is hindered by the lack of comprehensive CoT datasets. Current resources often fail to provide extensive reasoning problems with coherent CoT processes distilled from multiple teacher models and do not account for multifaceted properties describing the internal characterist"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10937","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-16T07:15:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"32d5d50fa4f233ac490f8b7891c645fa64f61c2c380af28ef5b2887920345d41","abstract_canon_sha256":"bbe1fc36a52220d005125df5b75f7662d7424863e46cda79a2d34d4158c8963d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:08.198253Z","signature_b64":"JR5aVKpAkvtjU3qmluoJjmHMg9CkhXM8mq5pdn9Gzd6XdsGO0o9LHdkoN27zzCGjLWKxeqwgtQHythaQm8zRBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a10f8c6cec46a9920dee874ec33178a49a4ec62a57210ce25c2a5bec206c9637","last_reissued_at":"2026-07-05T11:04:08.197849Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:08.197849Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reasoning with OmniThought: A Large CoT Dataset with Verbosity and Cognitive Difficulty Annotations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chengyu Wang, Junbing Yan, Jun Huang, Wenrui Cai, Xiangzhong Fang","submitted_at":"2025-05-16T07:15:30Z","abstract_excerpt":"The emergence of large reasoning models (LRMs) has transformed Natural Language Processing by excelling in complex tasks such as mathematical problem-solving and code generation. These models leverage chain-of-thought (CoT) processes, enabling them to emulate human-like reasoning strategies. However, the advancement of LRMs is hindered by the lack of comprehensive CoT datasets. Current resources often fail to provide extensive reasoning problems with coherent CoT processes distilled from multiple teacher models and do not account for multifaceted properties describing the internal characterist"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10937","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10937","created_at":"2026-07-05T11:04:08.197909+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10937v1","created_at":"2026-07-05T11:04:08.197909+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10937","created_at":"2026-07-05T11:04:08.197909+00:00"},{"alias_kind":"pith_short_12","alias_value":"UEHYY3HMI2UZ","created_at":"2026-07-05T11:04:08.197909+00:00"},{"alias_kind":"pith_short_16","alias_value":"UEHYY3HMI2UZEDPO","created_at":"2026-07-05T11:04:08.197909+00:00"},{"alias_kind":"pith_short_8","alias_value":"UEHYY3HM","created_at":"2026-07-05T11:04:08.197909+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26671","citing_title":"NebulaExp-8B: An Empirical Post-Training Pipeline via Full-Scale Ablation Research","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":294,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11629","citing_title":"OmniThoughtVis: A Scalable Distillation Pipeline for Deployable Multimodal Reasoning Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10480","citing_title":"Tracing the Roots: A Multi-Agent Framework for Uncovering Data Lineage in Post-Training LLMs","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS","json":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS.json","graph_json":"https://pith.science/api/pith-number/UEHYY3HMI2UZEDPOQ5HMGMLYUS/graph.json","events_json":"https://pith.science/api/pith-number/UEHYY3HMI2UZEDPOQ5HMGMLYUS/events.json","paper":"https://pith.science/paper/UEHYY3HM"},"agent_actions":{"view_html":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS","download_json":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS.json","view_paper":"https://pith.science/paper/UEHYY3HM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10937&json=true","fetch_graph":"https://pith.science/api/pith-number/UEHYY3HMI2UZEDPOQ5HMGMLYUS/graph.json","fetch_events":"https://pith.science/api/pith-number/UEHYY3HMI2UZEDPOQ5HMGMLYUS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS/action/storage_attestation","attest_author":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS/action/author_attestation","sign_citation":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS/action/citation_signature","submit_replication":"https://pith.science/pith/UEHYY3HMI2UZEDPOQ5HMGMLYUS/action/replication_record"}},"created_at":"2026-07-05T11:04:08.197909+00:00","updated_at":"2026-07-05T11:04:08.197909+00:00"}