{"as_of":"2026-08-05T20:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:40e3c4324b128fdbcb08c30676b816e7d9198ec5aff3c15d21dc163022186a53","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":27,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T16:19:53.921525Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-03T03:23:56.857266Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"reference_index":191,"source":"pdf_text","source_observed_at":"2026-05-14T01:29:56.480020Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2503.16419"},"observation_digest":"sha256:c60fe29fb2e56db2ab477381d88847675aed3d9cefd714717f4134ba7209d518","observation_id":"3a06483f-e30c-4c86-a0d8-fc1f430aa4c1","resolution":{"observed_at":"2026-05-14T01:29:57.071592Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2504.14945","last_updated":"2025-06-22T00:18:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-21T08:09:13Z","title":"Learning to Reason under Off-Policy Guidance","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-15T23:17:02.701393Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2504.14945"},"observation_digest":"sha256:207bc471391a0c4ce369b7a87a648ff8df02946d1fe817cfc3300bf8424634ed","observation_id":"616e83c4-3352-43ef-b983-289c2dbd6c2a","resolution":{"observed_at":"2026-05-15T23:17:02.869340Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-15T19:51:04.779597Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2504.20571"},"observation_digest":"sha256:48e52bd26d6097f11dbcbba42151e45640d652b60e8919705780446f62208931","observation_id":"89e88595-6b82-4a51-ba4e-fabbdd4e0f50","resolution":{"observed_at":"2026-05-15T19:51:05.002577Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2505.22312","last_updated":"2025-05-29T09:07:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-28T12:56:04Z","title":"Skywork Open Reasoner 1 Technical Report","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T04:26:47.283983Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2505.22312"},"observation_digest":"sha256:c3ba10bb9f8142cbfe51ad554e296e7b5d3774f4f3770528ee7b040e33003ec0","observation_id":"d906625e-03ee-4bc7-90cf-733794c7e472","resolution":{"observed_at":"2026-05-17T04:26:47.350497Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2508.08636","last_updated":"2026-05-20T08:06:05Z","snapshot_observed_at":"2026-08-04T03:35:42.053775Z","submitted_at":"2025-08-12T05:00:00Z","title":"InternBootcamp Technical Report: Boosting LLM Reasoning with Verifiable Task Scaling","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-21T22:33:09.674822Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2508.08636"},"observation_digest":"sha256:a84c9eeb03b768b8d35427991b41124e8059e1f4afca45fe71b90ec2900e0ef4","observation_id":"ba84593e-9ae9-4427-8d11-3e07adbfd0b4","resolution":{"observed_at":"2026-05-21T22:34:23.988237Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T16:19:53.921525Z","title":"Light-r1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.18773","last_updated":"2025-08-26T07:57:28Z","snapshot_observed_at":"2026-08-05T16:19:52.959116Z","submitted_at":"2025-08-26T07:57:28Z","title":"ThinkDial: An Open Recipe for Controlling Reasoning Effort in Large Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T16:19:53.921525Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2508.18773"},"observation_digest":"sha256:cf0fda3503db09555a01d5bdee9bf73931cd08fb681bf0c1a0b5cd88e6b242a4","observation_id":"c5b6fc67-3e29-4c6a-bd9e-45e90dcbf817","resolution":{"observed_at":"2026-08-05T16:19:53.921525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T15:11:01.110314Z","title":"arXiv:2503.10460","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.20384","last_updated":"2025-08-28T03:16:15Z","snapshot_observed_at":"2026-08-05T15:11:00.806701Z","submitted_at":"2025-08-28T03:16:15Z","title":"Uncertainty Under the Curve: A Sequence-Level Entropy Area Metric for Reasoning LLM","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T15:11:01.110314Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2508.20384"},"observation_digest":"sha256:3f703df1430c2bda905d9354c1c57d36ed24bcbb7347e9902cdcc1069a87ff8e","observation_id":"d205813b-567a-43ec-9793-c23f36ab7807","resolution":{"observed_at":"2026-08-05T15:11:01.110314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T05:45:04.041310Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.05007","last_updated":"2025-09-08T03:26:03Z","snapshot_observed_at":"2026-08-05T05:45:00.604843Z","submitted_at":"2025-09-05T11:14:11Z","title":"Sticker-TTS: Learn to Utilize Historical Experience with a Sticker-driven Test-Time Scaling Framework","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T05:45:04.041310Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.05007"},"observation_digest":"sha256:0e238f5ac0644d3f0aca06d6c07505a58425d502fa0684940a23aee1d2efc78f","observation_id":"08fe2101-9ae3-4671-9a47-74cd828efa0a","resolution":{"observed_at":"2026-08-05T05:45:04.041310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T05:45:04.085149Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.05007","last_updated":"2025-09-08T03:26:03Z","snapshot_observed_at":"2026-08-05T05:45:00.604843Z","submitted_at":"2025-09-05T11:14:11Z","title":"Sticker-TTS: Learn to Utilize Historical Experience with a Sticker-driven Test-Time Scaling Framework","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T05:45:04.085149Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.05007"},"observation_digest":"sha256:9100902eb238aa9c1fc70c0ff2019c464f7d2feb7f8514b120580132202741d8","observation_id":"2df28670-0e00-44f9-a3c3-8772a2db9d8a","resolution":{"observed_at":"2026-08-05T05:45:04.085149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-04T23:23:41.779153Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.06650","last_updated":"2025-09-08T13:04:07Z","snapshot_observed_at":"2026-08-04T23:23:38.041452Z","submitted_at":"2025-09-08T13:04:07Z","title":"Domain-Aware RAG: MoL-Enhanced RL for Efficient Training and Scalable Retrieval","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T23:23:41.779153Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.06650"},"observation_digest":"sha256:96851079a7782415ddcf50f834ec28f22922eae9405495f89bd33188b2d193a5","observation_id":"e1c883b7-520d-4b4a-97f1-d8b690a827bc","resolution":{"observed_at":"2026-08-04T23:23:41.779153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2509.25758","last_updated":"2026-04-14T15:45:09Z","snapshot_observed_at":"2026-08-04T09:27:32.441799Z","submitted_at":"2025-09-30T04:23:43Z","title":"Thinking Sparks!: Emergent Attention Heads in Reasoning Models During Post Training","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-18T13:28:32.093512Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2509.25758"},"observation_digest":"sha256:b21677f8529e15a76ed588caaf4ac2d0a4048c463669c1a356ae02655c8abd3d","observation_id":"400fda6c-c35b-4b51-92e0-b5081c768edb","resolution":{"observed_at":"2026-05-18T13:31:24.978441Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.03988","last_updated":"2026-04-14T20:09:39Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T01:15:32Z","title":"The Signal is in the Steps: Local Scoring for Reasoning Data Selection","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T10:14:27.739531Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.03988"},"observation_digest":"sha256:f51aac2118dac7aca7c0ffa14a4b0cf5ddd8a201619bf7673d63b57da2997a63","observation_id":"b7b7f24b-3eea-4265-9b27-09e63c6e8bb5","resolution":{"observed_at":"2026-05-18T10:16:14.172268Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.04265","last_updated":"2026-05-12T01:55:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-05T16:14:03Z","title":"Don't Pass@k: A Bayesian Framework for Large Language Model Evaluation","version":4},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-05-18T10:04:39.223895Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.04265"},"observation_digest":"sha256:4a3be252b1c8a8c4e0a3c027cd9aa44726aa6ceadcef49391a3749aeec70d122","observation_id":"6eb96528-8a20-4626-b3e4-0e81d6459750","resolution":{"observed_at":"2026-05-18T10:06:13.802986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.18814","last_updated":"2026-07-05T03:44:20Z","snapshot_observed_at":"2026-08-04T08:53:03.599438Z","submitted_at":"2025-10-21T17:15:56Z","title":"A Model Can Help Itself: Reward-Free Self-Training for LLM Reasoning","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T05:11:29.205366Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.18814"},"observation_digest":"sha256:0f782cbd0c55d0ea650ca6e6e022b4350852288eb57bd38b02a4fb8b0b9e4435","observation_id":"15289bb2-e131-49f9-ae78-6e8beb41fd51","resolution":{"observed_at":"2026-05-18T05:12:23.662328Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2510.18814","last_updated":"2026-07-05T03:44:20Z","snapshot_observed_at":"2026-08-04T08:53:03.599438Z","submitted_at":"2025-10-21T17:15:56Z","title":"A Model Can Help Itself: Reward-Free Self-Training for LLM Reasoning","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T20:06:16.172916Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.18814"},"observation_digest":"sha256:1df9dba8c6af3416d3e7d1c641cfef1d6f05d698a47a3a5197b7939d162372ba","observation_id":"be7e2720-b16e-4fdb-ae74-588e4da07b09","resolution":{"observed_at":"2026-05-21T20:10:34.891829Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-04T08:53:07.747822Z","title":"Light-R1: Curriculum sft, dpo and rl for long cot from scratch and beyond.arXiv preprint arXiv:2503.10460,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.18814","last_updated":"2026-07-05T03:44:20Z","snapshot_observed_at":"2026-08-04T08:53:03.599438Z","submitted_at":"2025-10-21T17:15:56Z","title":"A Model Can Help Itself: Reward-Free Self-Training for LLM Reasoning","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T08:53:07.747822Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2510.18814"},"observation_digest":"sha256:4d1b6057cdac738d4657e80f81aa7d7633e3795ff797a8ee19a612d7f82ace42","observation_id":"a4abdab7-eff5-47e4-8935-3ee46b435881","resolution":{"observed_at":"2026-08-04T08:53:07.747822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2512.03847","last_updated":"2026-05-06T14:15:19Z","snapshot_observed_at":"2026-08-03T04:49:11.537109Z","submitted_at":"2025-12-03T14:48:38Z","title":"DVPO: Distributional Value Modeling-based Policy Optimization for LLM Post-Training","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T01:46:21.744857Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2512.03847"},"observation_digest":"sha256:9a4605af5cf6a64f9c33866f0fd6d11b4b2ce382f3d7070ab378223d21d244f3","observation_id":"d14952ea-21c3-4e8c-8ef4-34e0280d2194","resolution":{"observed_at":"2026-05-17T01:48:50.926478Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2602.01970","last_updated":"2026-05-15T12:23:06Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T11:24:36Z","title":"Small Generalizable Prompt Predictive Models Can Steer Efficient RL Post-Training of Large Reasoning Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-21T14:09:26.842696Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2602.01970"},"observation_digest":"sha256:c798a81c09be28d99a7a788d86669d89ef8e396c23c07cbf9576ef234d5d03f1","observation_id":"1bf0c8de-d4d5-449c-998f-c000e8621a0d","resolution":{"observed_at":"2026-05-21T14:10:13.173355Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2604.17614","last_updated":"2026-04-19T20:58:25Z","snapshot_observed_at":"2026-08-01T22:49:57.966435Z","submitted_at":"2026-04-19T20:58:25Z","title":"Characterizing Model-Native Skills","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-10T05:42:49.694715Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2604.17614"},"observation_digest":"sha256:cd3659979ac009d2deb789353b5871715e2269572e5161f2097d9b1c2eea2eb5","observation_id":"09a06dac-043b-4ecc-bfb4-0a1933af2e20","resolution":{"observed_at":"2026-05-10T06:06:19.521433Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.06165","last_updated":"2026-05-07T12:51:49Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T12:51:49Z","title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","version":1},"reference_index":159,"source":"arxiv_source","source_observed_at":"2026-05-08T10:19:08.451445Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.06165"},"observation_digest":"sha256:4d44d1a081ad6971b1656d7126236b6176c2ccc23c3e4f779b063ebb7139ea53","observation_id":"0e9be8f2-6c0b-4de7-98bb-6803cffdb2a4","resolution":{"observed_at":"2026-05-11T20:06:09.755531Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.08441","last_updated":"2026-05-08T20:03:19Z","snapshot_observed_at":"2026-08-02T05:33:25.541249Z","submitted_at":"2026-05-08T20:03:19Z","title":"DUET: Optimize Token-Budget Allocation for Reinforcement Learning with Verifiable Rewards","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-12T01:57:11.065744Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.08441"},"observation_digest":"sha256:7d7aa00aade192796fe32b90b7c8f517aafcc64e714383a2c9cbf01db82c9428","observation_id":"5a49ae54-1398-4a86-9a44-d95a03d6c9a1","resolution":{"observed_at":"2026-05-12T07:46:28.509523Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.08905","last_updated":"2026-05-09T11:57:25Z","snapshot_observed_at":"2026-07-06T23:21:06.773761Z","submitted_at":"2026-05-09T11:57:25Z","title":"Forge: Quality-Aware Reinforcement Learning for NP-Hard Optimization in LLMs","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-12T02:44:33.143247Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.08905"},"observation_digest":"sha256:5f41ffc8ecadc13d7fe22295fb28c941c27ef3b0a0643e03a55d3c78f72ea6ae","observation_id":"e229cccd-8685-440f-a136-6c54e5289bd3","resolution":{"observed_at":"2026-05-12T02:46:18.839590Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.14054","last_updated":"2026-06-02T19:43:46Z","snapshot_observed_at":"2026-07-06T23:25:29.856145Z","submitted_at":"2026-05-13T19:23:53Z","title":"Bad Seeing or Bad Thinking? Rewarding Perception for Multimodal Reasoning","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-15T05:14:28.256032Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.14054"},"observation_digest":"sha256:56681098f75bac1efd9d499001af6e050aef3d06fe23255bf7a5c942a2989537","observation_id":"1d6efd00-eb3d-4326-8317-629164fefbbe","resolution":{"observed_at":"2026-05-15T05:15:02.865013Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2605.22567","last_updated":"2026-05-21T14:47:52Z","snapshot_observed_at":"2026-07-06T23:32:54.127514Z","submitted_at":"2026-05-21T14:47:52Z","title":"LANG: Reinforcement Learning for Multilingual Reasoning with Language-Adaptive Hint Guidance","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-22T06:19:44.377733Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2605.22567"},"observation_digest":"sha256:e90e5c5ca4181db4a814ac288f8c6e4496421da30d850dbc88187d6a778f1f81","observation_id":"d4a5b6d9-9cbb-40e4-9b29-12e08523d53f","resolution":{"observed_at":"2026-05-22T06:21:09.548667Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2606.01168","last_updated":"2026-05-31T11:20:00Z","snapshot_observed_at":"2026-08-01T09:00:59.914837Z","submitted_at":"2026-05-31T11:20:00Z","title":"Thinking Economically: A Hierarchical Framework for Adaptive-Complexity Reasoning in LLMs","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-28T17:05:48.244094Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2606.01168"},"observation_digest":"sha256:2ce68f3ab965e7c054e70e8410e9104cd0e58b6fb6d6a4577aa8539a1ec3c52a","observation_id":"ae67e4f2-82f4-4c79-bc4f-db0a819cbd93","resolution":{"observed_at":"2026-06-28T17:12:25.200187Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2606.11119","last_updated":"2026-06-09T17:16:03Z","snapshot_observed_at":"2026-07-31T16:22:30.560541Z","submitted_at":"2026-06-09T17:16:03Z","title":"TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-27T13:55:35.363377Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2606.11119"},"observation_digest":"sha256:2a836686e29e31c0c2b652e27761625c32fe6271498c0d8a4a948a0d774a9917","observation_id":"221004fe-29e5-487f-8700-2e681c0208f9","resolution":{"observed_at":"2026-07-03T04:27:36.717007Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","version":4},"cited_work":{"arxiv_id":"2503.10460","doi":"10.48550/arxiv.2503.10460","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.10460","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"arXiv (Cornell University)","work_id":"83fc58bb-3060-4d01-ac57-a76ad19c36d6","year":2025},"citing_paper":{"arxiv_id":"2607.02234","last_updated":"2026-07-02T14:33:07Z","snapshot_observed_at":"2026-07-07T00:07:42.752664Z","submitted_at":"2026-07-02T14:33:07Z","title":"Purified OPSD: On-Policy Self-Distillation Without Losing How to Think","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-03T13:56:13.827493Z"},"links":{"cited_paper":"/paper/2503.10460","citing_paper":"/paper/2607.02234"},"observation_digest":"sha256:5bec3e04c3c24bcfda406a7378f93c4b8600deca74c8e4fbdd9f74db09d75a31","observation_id":"ec1f147d-143b-4586-bbec-816290c2e37b","resolution":{"observed_at":"2026-07-03T13:58:21.051491Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2503.10460/citation-record","integrity":"/paper/2503.10460/integrity","json":"/paper/2503.10460/citation-record.json","paper":"/paper/2503.10460"},"outbound":[],"paper":{"arxiv_id":"2503.10460","last_updated":"2025-05-28T12:32:29Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T20:52:04.228632Z","submitted_at":"2025-03-13T15:29:22Z","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 27 inbound Pith citation observations for arXiv:2503.10460."}