{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UFQ4DJRIZSUHG5FIOYX4ZTNDF6","short_pith_number":"pith:UFQ4DJRI","schema_version":"1.0","canonical_sha256":"a161c1a628cca87374a8762fcccda32fb8eed647e56f0d8fa72afe42f644bc98","source":{"kind":"arxiv","id":"2510.09278","version":2},"attestation_state":"computed","paper":{"title":"CLARity: Reasoning Consistency Alone Can Teach Reinforced Experts","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cong Jiang, Jiarui Sun, Jiuheng Lin, Yansong Feng, Zirui Wu","submitted_at":"2025-10-10T11:21:09Z","abstract_excerpt":"Training expert LLMs in domains with scarce data is difficult, often relying on multiple-choice questions (MCQs). However, standard outcome-based reinforcement learning (RL) on MCQs is risky. While it may improve accuracy, we observe it often degrades reasoning quality such as logical consistency. Existing solutions to supervise reasoning, such as large-scale Process Reward Models (PRMs), are prohibitively expensive. To address this, we propose CLARity, a cost-effective RL framework that enhances reasoning quality using only a small, general-purpose LLM. CLARity integrates a consistency-aware "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.09278","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-10-10T11:21:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0f6ba4bc8d72336888f1530ca75f8624dd97c53fcd1f43f246ec2448d8bb6280","abstract_canon_sha256":"75179b308a40dba6aebd7de7ef3b7edae9c6433af8f7f2c884d590a7c95e33e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-30T01:17:27.515588Z","signature_b64":"DSofw9OyFW+RzS9HveNWn6kDOkxilrkzPNy2IkXmDjnboAzK+7GtI0iXl5ODj+CPezErX1rdKGrVwg3pzSiIAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a161c1a628cca87374a8762fcccda32fb8eed647e56f0d8fa72afe42f644bc98","last_reissued_at":"2026-06-30T01:17:27.514937Z","signature_status":"signed_v1","first_computed_at":"2026-06-30T01:17:27.514937Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLARity: Reasoning Consistency Alone Can Teach Reinforced Experts","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cong Jiang, Jiarui Sun, Jiuheng Lin, Yansong Feng, Zirui Wu","submitted_at":"2025-10-10T11:21:09Z","abstract_excerpt":"Training expert LLMs in domains with scarce data is difficult, often relying on multiple-choice questions (MCQs). However, standard outcome-based reinforcement learning (RL) on MCQs is risky. While it may improve accuracy, we observe it often degrades reasoning quality such as logical consistency. Existing solutions to supervise reasoning, such as large-scale Process Reward Models (PRMs), are prohibitively expensive. To address this, we propose CLARity, a cost-effective RL framework that enhances reasoning quality using only a small, general-purpose LLM. CLARity integrates a consistency-aware "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.09278","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.09278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.09278","created_at":"2026-06-30T01:17:27.515014+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.09278v2","created_at":"2026-06-30T01:17:27.515014+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.09278","created_at":"2026-06-30T01:17:27.515014+00:00"},{"alias_kind":"pith_short_12","alias_value":"UFQ4DJRIZSUH","created_at":"2026-06-30T01:17:27.515014+00:00"},{"alias_kind":"pith_short_16","alias_value":"UFQ4DJRIZSUHG5FI","created_at":"2026-06-30T01:17:27.515014+00:00"},{"alias_kind":"pith_short_8","alias_value":"UFQ4DJRI","created_at":"2026-06-30T01:17:27.515014+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6","json":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6.json","graph_json":"https://pith.science/api/pith-number/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/graph.json","events_json":"https://pith.science/api/pith-number/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/events.json","paper":"https://pith.science/paper/UFQ4DJRI"},"agent_actions":{"view_html":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6","download_json":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6.json","view_paper":"https://pith.science/paper/UFQ4DJRI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.09278&json=true","fetch_graph":"https://pith.science/api/pith-number/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/graph.json","fetch_events":"https://pith.science/api/pith-number/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/action/storage_attestation","attest_author":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/action/author_attestation","sign_citation":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/action/citation_signature","submit_replication":"https://pith.science/pith/UFQ4DJRIZSUHG5FIOYX4ZTNDF6/action/replication_record"}},"created_at":"2026-06-30T01:17:27.515014+00:00","updated_at":"2026-06-30T01:17:27.515014+00:00"}