{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NKFJQDTMOVTBHZU5SQV5POY4LO","short_pith_number":"pith:NKFJQDTM","schema_version":"1.0","canonical_sha256":"6a8a980e6c756613e69d942bd7bb1c5b851b4eac9396b4018ef881ff40937a69","source":{"kind":"arxiv","id":"2501.11651","version":2},"attestation_state":"computed","paper":{"title":"T1: Advancing Language Model Reasoning through Reinforcement Learning and Inference Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jiajie Zhang, Jie Tang, Juanzi Li, Rui Lu, Xin Lv, Yujiang Li, Yuxiao Dong, Zhenyu Hou, Zijun Yao","submitted_at":"2025-01-20T18:33:33Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in complex reasoning tasks. However, existing approaches mainly rely on imitation learning and struggle to achieve effective test-time scaling. While reinforcement learning (RL) holds promise for enabling self-exploration, recent attempts yield modest improvements in complex reasoning. In this paper, we present T1 to scale RL by encouraging exploration and understand inference scaling. We first initialize the LLM using synthesized chain-of-thought data that integrates trial-and-error and self-verification. To scale RL train"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.11651","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-20T18:33:33Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f28df3ee434cbda6af1645b07025d1f223d5425f4019ac8111a97cfb28b63f0d","abstract_canon_sha256":"81b3fb41acecd9f74f2fd42a2728a8ca717380f4b45fc230e658f500327a6f03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:49.856172Z","signature_b64":"GxE+K5jRyDnhWmhAZnTeY5I4X4HxcQZ6IpopJDXuutf/l4tNHEyrCBmxwG6e81Y/c/krL1MDU8w7bb3+fNQoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a8a980e6c756613e69d942bd7bb1c5b851b4eac9396b4018ef881ff40937a69","last_reissued_at":"2026-07-05T11:20:49.855663Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:49.855663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"T1: Advancing Language Model Reasoning through Reinforcement Learning and Inference Scaling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jiajie Zhang, Jie Tang, Juanzi Li, Rui Lu, Xin Lv, Yujiang Li, Yuxiao Dong, Zhenyu Hou, Zijun Yao","submitted_at":"2025-01-20T18:33:33Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable capabilities in complex reasoning tasks. However, existing approaches mainly rely on imitation learning and struggle to achieve effective test-time scaling. While reinforcement learning (RL) holds promise for enabling self-exploration, recent attempts yield modest improvements in complex reasoning. In this paper, we present T1 to scale RL by encouraging exploration and understand inference scaling. We first initialize the LLM using synthesized chain-of-thought data that integrates trial-and-error and self-verification. To scale RL train"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11651","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.11651","created_at":"2026-07-05T11:20:49.855734+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.11651v2","created_at":"2026-07-05T11:20:49.855734+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11651","created_at":"2026-07-05T11:20:49.855734+00:00"},{"alias_kind":"pith_short_12","alias_value":"NKFJQDTMOVTB","created_at":"2026-07-05T11:20:49.855734+00:00"},{"alias_kind":"pith_short_16","alias_value":"NKFJQDTMOVTBHZU5","created_at":"2026-07-05T11:20:49.855734+00:00"},{"alias_kind":"pith_short_8","alias_value":"NKFJQDTM","created_at":"2026-07-05T11:20:49.855734+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05861","citing_title":"Mitigating Factual Hallucination in Large Reasoning Models via Mixed-Mode Advantage Regularization","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2510.08141","citing_title":"SCOPE-RL: Stable and Quantitative Control of Policy Entropy in RL Post-Training","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11461","citing_title":"Breaking $\\textit{Winner-Takes-All}$: Cooperative Policy Optimization Improves Diverse LLM Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18864","citing_title":"SAGE: Shaping Anchors for Guided Exploration in RLVR of LLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05837","citing_title":"EEPO: Exploration-Enhanced Policy Optimization via Sample-Then-Forget","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02283","citing_title":"Self-Forcing++: Towards Minute-Scale High-Quality Video Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11461","citing_title":"Breaking $\\textit{Winner-Takes-All}$: Cooperative Policy Optimization Improves Diverse LLM Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":269,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14646","citing_title":"Targeted Exploration via Unified Entropy Control for Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17433","citing_title":"Self-Consistency from Only Two Samples: CoT-PoT Ensembling for Efficient LLM Reasoning","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO","json":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO.json","graph_json":"https://pith.science/api/pith-number/NKFJQDTMOVTBHZU5SQV5POY4LO/graph.json","events_json":"https://pith.science/api/pith-number/NKFJQDTMOVTBHZU5SQV5POY4LO/events.json","paper":"https://pith.science/paper/NKFJQDTM"},"agent_actions":{"view_html":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO","download_json":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO.json","view_paper":"https://pith.science/paper/NKFJQDTM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.11651&json=true","fetch_graph":"https://pith.science/api/pith-number/NKFJQDTMOVTBHZU5SQV5POY4LO/graph.json","fetch_events":"https://pith.science/api/pith-number/NKFJQDTMOVTBHZU5SQV5POY4LO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/action/storage_attestation","attest_author":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/action/author_attestation","sign_citation":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/action/citation_signature","submit_replication":"https://pith.science/pith/NKFJQDTMOVTBHZU5SQV5POY4LO/action/replication_record"}},"created_at":"2026-07-05T11:20:49.855734+00:00","updated_at":"2026-07-05T11:20:49.855734+00:00"}