{"as_of":"2026-08-05T08:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:35e9d424412b30f4e1e78db5c10ef9f19d80c7a72bc3d65f63d3ea348778911b","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-23T02:41:21.571824Z","state":"measured"},{"denominator":90,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":90,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T04:17:27.154486Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12272","snapshot_observed_at":"2026-08-03T04:17:27.154486Z","title":"Learning to reason at the frontier of learnability.arXiv preprint arXiv:2502.12272, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05547","last_updated":"2026-07-27T11:11:55Z","snapshot_observed_at":"2026-08-05T06:35:43.360705Z","submitted_at":"2026-02-05T11:06:37Z","title":"Multi-Task GRPO: Reliable LLM Reasoning Across Tasks","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T04:17:27.154486Z"},"links":{"cited_paper":"/paper/2502.12272","citing_paper":"/paper/2602.05547"},"observation_digest":"sha256:49a5f87546c9d2fe4bd8b90d1a8d80f542ad67217c6f255a5fc65052282134a4","observation_id":"2e123e76-4eb5-44c5-8243-1d08c19a53f0","resolution":{"observed_at":"2026-08-03T04:17:27.154486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.12272/citation-record","integrity":"/paper/2502.12272/integrity","json":"/paper/2502.12272/citation-record.json","paper":"/paper/2502.12272"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:951b847ca46b2c0a9214db02a4ce8dead0cae4912243b9df23f87b9d666fb9d6","observation_id":"4f7f7a4c-d5f1-4184-8a17-8f69d175f469","resolution":{"observed_at":"2026-05-23T02:42:26.069190Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-07-06T19:55:37.400185Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":"2411.15124","doi":"10.48550/arxiv.2411.15124","metadata_source":"pith","pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","venue":"cs.CL","work_id":"28c9dbea-056a-48c2-8000-85f809827e45","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:c6063acee2c8bbe30e3bcf4faac8ed4e4e8c0bf5ace95d57d88cecfac40a4abd","observation_id":"c2cbddbf-206a-4b79-9e27-6848eb81e78e","resolution":{"observed_at":"2026-05-23T02:42:26.038058Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to reason with llms","venue":null,"work_id":"bae5581f-90bc-49ab-b0f8-43401921dfca","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:89a21fd564fe8b2376ec34ad3cb2cd0a723a6ea82edc46395f0bb2fe35a911ce","observation_id":"fdaf63a6-feec-45ff-9f96-01aa0bad26f3","resolution":{"observed_at":"2026-05-23T02:47:27.556981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":"2503.20783","doi":"10.48550/arxiv.2503.20783","metadata_source":"pith","pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","venue":"cs.LG","work_id":"ec354f3b-9484-4a0c-94c8-92d4d0260835","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:18ec8099204a5630239ccdf17e9874527bbe6ecc001c0b1989235c7e8598aa05","observation_id":"b290130e-c07f-4672-9a7d-35b03ff5ed0b","resolution":{"observed_at":"2026-05-23T02:42:26.030208Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vineppo: Unlocking rl potential for llm reasoning through refined credit assignment","venue":null,"work_id":"01c63620-3179-441e-a7bd-a39be4b564b8","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:3c9141b3c7e4726ea3e8acee015d5d52c2fc4bbd9d1ef8593558958ceba94e7b","observation_id":"7046d60f-b54a-4c5a-a42b-cca48dcecaad","resolution":{"observed_at":"2026-05-23T02:47:27.553779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01679","last_updated":"2025-06-03T20:51:06Z","snapshot_observed_at":"2026-07-06T19:26:12.102947Z","submitted_at":"2024-10-02T15:49:30Z","title":"VinePPO: Refining Credit Assignment in RL Training of LLMs","version":2},"cited_work":{"arxiv_id":"2410.01679","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.01679","snapshot_observed_at":"2026-07-09T11:16:11.296273Z","title":"Available: https://arxiv.org/abs/2410.01679","venue":"cs.LG","work_id":"1b370c79-344d-4e23-8065-f313dbdc84de","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2410.01679","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:ab9dd9f44cac143dbfc4f047e9cc95ef3002f4f7fb9912f9942b4e704f70504f","observation_id":"66fa54c1-421c-4938-8538-0c19b137cf03","resolution":{"observed_at":"2026-05-23T02:42:26.025966Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20304","last_updated":"2024-05-30T17:50:04Z","snapshot_observed_at":"2026-08-04T05:00:56.526394Z","submitted_at":"2024-05-30T17:50:04Z","title":"Group Robust Preference Optimization in Reward-free RLHF","version":1},"cited_work":{"arxiv_id":"2405.20304","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.20304","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Group robust preference optimization in reward-free rlhf","venue":null,"work_id":"c3ea3ccd-d175-4914-95b0-9b717591d6af","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2405.20304","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:decc02958c4d82831b01ff096552af10081dff02bd6a8856bcf654a1db177935","observation_id":"79f22151-c393-4174-970b-f15eaa667fd7","resolution":{"observed_at":"2026-05-23T02:42:26.236094Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:4a9627cbef3e7e65f9853f010dc5849ba2fd8b9e0d15588f8e75c9edf4c22845","observation_id":"b2f396c2-77ad-4b05-8ab4-ceb0fcc0b99c","resolution":{"observed_at":"2026-05-23T02:42:26.231736Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-04T15:46:25.710484Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:06eff0bd81e0c19e1157d084be52271daf78cfe18bb1c23cc3dc0ba9eed0d1e8","observation_id":"137dd64d-fe7f-46dd-bbdb-6ecd5ba5c84e","resolution":{"observed_at":"2026-05-23T02:42:26.227411Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2fcc4e27c6ab95215b659b8f9b944b96cf97765e4d51eefb6f04b16a9099d10a","observation_id":"42f79281-0270-42b9-9207-899617751327","resolution":{"observed_at":"2026-05-23T02:42:26.081240Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24290","last_updated":"2025-07-05T09:01:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-31T16:36:05Z","title":"Open-Reasoner-Zero: An Open Source Approach to Scaling Up Reinforcement Learning on the Base Model","version":2},"cited_work":{"arxiv_id":"2503.24290","doi":"10.48550/arxiv.2503.24290","metadata_source":"pith","pith_arxiv_id":"2503.24290","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Open-Reasoner-Zero: An Open Source Approach to Scaling Up Reinforcement Learning on the Base Model","venue":"cs.LG","work_id":"763e0e44-40dd-4bdd-8414-21f8f9ce6d10","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2503.24290","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:06961f4348797e2bc670c31b53fa590e2700acdd01589b62f3b35b791a50600a","observation_id":"0c878d87-271b-49f9-a748-107161b0410d","resolution":{"observed_at":"2026-05-23T02:42:26.222795Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07965","last_updated":"2025-01-08T09:07:54Z","snapshot_observed_at":"2026-07-06T17:59:00.124173Z","submitted_at":"2024-04-11T17:52:01Z","title":"Rho-1: Not All Tokens Are What You Need","version":4},"cited_work":{"arxiv_id":"2404.07965","doi":"10.48550/arxiv.2404.07965","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.07965","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rho-1: Not all tokens are what you need","venue":"arXiv (Cornell University)","work_id":"74e3da93-f1b3-41df-bc52-f8092041fb58","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2404.07965","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:cd6119ef0f1f95314215817772a6fc5ba51ad6a86e49baa9f08c871639c8ee7a","observation_id":"95ee720f-f476-4f9e-b569-0f3e85e7c30c","resolution":{"observed_at":"2026-05-23T02:42:26.218182Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Qwen2.5 technical report","venue":null,"work_id":"71f8c882-c442-4947-ae66-b3442d2519b1","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:208db552e4cbd859addc37a3c793abf8cb5cce3cbcd3e6b45296bbdb9f363583","observation_id":"f955939b-2b2a-4736-afb8-5f09842eb4d9","resolution":{"observed_at":"2026-05-23T02:47:27.549825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.02884","last_updated":"2024-03-05T11:42:59Z","snapshot_observed_at":"2026-07-06T17:39:49.484077Z","submitted_at":"2024-03-05T11:42:59Z","title":"MathScale: Scaling Instruction Tuning for Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":"2403.02884","doi":"10.48550/arxiv.2403.02884","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.02884","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mathscale: Scaling instruction tuning for mathematical reasoning","venue":"arXiv (Cornell University)","work_id":"450d5342-49ed-44d9-85e1-41826683418c","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2403.02884","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:4c631812ebff4a8a300629788b650adaf5cfd94df98ddbb23df2089f01897b3e","observation_id":"6a7030aa-b42a-4730-9853-349c3df3b941","resolution":{"observed_at":"2026-05-23T02:42:26.213553Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14008","last_updated":"2024-06-06T13:19:44Z","snapshot_observed_at":"2026-08-03T03:39:09.398343Z","submitted_at":"2024-02-21T18:49:26Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","version":2},"cited_work":{"arxiv_id":"2402.14008","doi":"10.48550/arxiv.2402.14008","metadata_source":"pith","pith_arxiv_id":"2402.14008","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","venue":"cs.CL","work_id":"19abed3b-0ff6-409b-aded-a50205319aa3","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2402.14008","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:0f55277089de51f2a705f3288158db772a52cfa828fd3ebcd02b24a4ef49676f","observation_id":"8393f775-1f3a-43ca-937d-3c5d42a98506","resolution":{"observed_at":"2026-05-23T02:42:26.208213Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-07-06T17:34:07.737296Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":"2402.14740","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-07-09T08:56:06.435604Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","venue":"cs.LG","work_id":"7bb8f9ec-1241-4472-a4fa-c636c6d79892","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:e898557e686637425786afdaf3e1315da3cfc8e8aef24c25d04bff01d133d2a7","observation_id":"e66f894e-8ce2-45d8-b1a6-37b0f8284d24","resolution":{"observed_at":"2026-05-23T02:42:26.045213Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12877","last_updated":"2023-04-25T14:49:34Z","snapshot_observed_at":"2026-07-06T15:19:50.076606Z","submitted_at":"2023-04-25T14:49:34Z","title":"Proximal Curriculum for Reinforcement Learning Agents","version":1},"cited_work":{"arxiv_id":"2304.12877","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.12877","snapshot_observed_at":"2026-07-04T18:30:01.557405Z","title":"Proximal curriculum for reinforcement learning agents","venue":null,"work_id":"3ce922ea-609e-413a-8854-eaf2fe8de7ab","year":2023},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2304.12877","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:985cccc828aaae8afd2d933e3a297d94e63fc152bf4e74d8be0cf22ade3bf3d8","observation_id":"f494981d-ec13-4c91-babe-835993ab72ce","resolution":{"observed_at":"2026-05-23T02:42:26.073171Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Automatic goal generation for reinforcement learning agents","venue":null,"work_id":"22b828a3-6d38-4cc8-acd3-39dd737d21b2","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:deed924a1eea8342fd4b00c53426705287789d8a54caab4371c117cc1d031a22","observation_id":"ebdc1395-d93e-4c1b-9d96-bd4315a448ef","resolution":{"observed_at":"2026-05-23T02:47:27.523015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15099","last_updated":"2024-10-29T18:25:44Z","snapshot_observed_at":"2026-07-06T19:06:37.442341Z","submitted_at":"2024-08-27T14:31:54Z","title":"No Regrets: Investigating and Improving Regret Approximations for Curriculum Discovery","version":3},"cited_work":{"arxiv_id":"2408.15099","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2408.15099","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"No regrets: Investigating and improving regret approximations for curriculum discovery","venue":null,"work_id":"52cfcb90-5946-49d9-9b3a-531c79b0a64b","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2408.15099","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:9d66f8bd852a4efaf626cf0d05d147864aad984c8249597df1688d72d1b5899b","observation_id":"1dce26a7-21f9-435d-a4e5-30ef10cb0337","resolution":{"observed_at":"2026-05-23T02:42:26.141497Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/bf00992696","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:47:34.373100Z","title":"Williams","venue":"Machine Learning","work_id":"469b3b81-55f9-4542-9dd7-570a63cfda74","year":1992},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:0949fa8e81be844e8f86866a21dd91fc130ae7d6a6b7d6c007f54db4bd7072fe","observation_id":"6e35180a-fdf5-4874-95a9-d1a9af850592","resolution":{"observed_at":"2026-05-23T02:42:25.514626Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-18T14:51:33.972563+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T14:51:33.972563+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.18252","last_updated":"2025-04-26T08:33:32Z","snapshot_observed_at":"2026-08-04T13:37:48.046113Z","submitted_at":"2024-10-23T19:59:50Z","title":"Asynchronous RLHF: Faster and More Efficient Off-Policy RL for Language Models","version":3},"cited_work":{"arxiv_id":"2410.18252","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.18252","snapshot_observed_at":"2026-07-10T21:57:39.335341Z","title":"Asynchronous rlhf: Faster and more efficient off-policy rl for language models","venue":"cs.LG","work_id":"89e577b6-ccb0-422f-9751-3ad83561b117","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2410.18252","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:5e74d1b4790b8cadf6c68cc6d73eec7ebb84f193cb4f464bf40f068d880707d6","observation_id":"dca9b697-bd7c-4ec7-9771-f0cff09a2db2","resolution":{"observed_at":"2026-05-23T02:42:26.179007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"cited_work":{"arxiv_id":"2405.11143","doi":"10.48550/arxiv.2405.11143","metadata_source":"pith","pith_arxiv_id":"2405.11143","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","venue":"cs.AI","work_id":"70fa48c9-2f84-49f6-9aca-37476e021fc3","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2405.11143","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:68349b4e0238d2291ce9aae1cbb286b091d1a84d61ed600d82887493b1f48e50","observation_id":"aaeb81a3-ee00-4047-b28f-a9f6e50632e0","resolution":{"observed_at":"2026-05-23T02:42:26.089482Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"There may not be aha moment in r1-zero-like training — a pilot study","venue":null,"work_id":"08ad3897-16ac-4982-bc95-5b9709d32f56","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:5d02b74d024bf0fc80fa441c7442dd51d44ec72c61bda1bd84eed81962d41bb6","observation_id":"10e0113a-ee17-49aa-9fcc-e8b7130c4ad7","resolution":{"observed_at":"2026-05-23T02:47:27.519337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Numinamath","venue":null,"work_id":"56dbd789-1e68-4f24-852a-4b84cf032a17","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:83deeb6972837376fe3a1509d29557cbdd58b787c4d75c4e63618fe2ad509e81","observation_id":"ee5c9610-11d4-4c2e-856c-651230236188","resolution":{"observed_at":"2026-05-23T02:47:27.573705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.14858","last_updated":"2022-07-01T02:15:12Z","snapshot_observed_at":"2026-07-06T13:26:06.337115Z","submitted_at":"2022-06-29T18:54:49Z","title":"Solving Quantitative Reasoning Problems with Language Models","version":2},"cited_work":{"arxiv_id":"2206.14858","doi":"10.48550/arxiv.2206.14858","metadata_source":"pith","pith_arxiv_id":"2206.14858","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Solving Quantitative Reasoning Problems with Language Models","venue":"cs.CL","work_id":"17214d12-1ca8-4186-806d-53c6715383a0","year":2022},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2206.14858","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a6aaa73549bced0341f566b9d2b9c391f86e26622ad04a1a389802252355dfaa","observation_id":"d8c04222-da3e-444d-b7b9-697610f088e4","resolution":{"observed_at":"2026-05-23T02:42:26.053039Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-21T13:53:28.356363+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T13:53:28.356363+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04642","last_updated":"2024-03-07T16:36:29Z","snapshot_observed_at":"2026-07-06T17:41:06.530605Z","submitted_at":"2024-03-07T16:36:29Z","title":"Teaching Large Language Models to Reason with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2403.04642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04642","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Teaching large language models to reason with reinforcement learning","venue":null,"work_id":"ab9d8347-574c-4fbb-b8e1-44eeba9c66b9","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2403.04642","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:f00d0b87af92684942e5ab8d48d2c5fad594e36a46ea4082909960f457641209","observation_id":"23390e2f-76d4-4c61-b3fb-2f047a4552c5","resolution":{"observed_at":"2026-05-23T02:42:26.085841Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.03934","last_updated":"2021-06-12T10:50:10Z","snapshot_observed_at":"2026-07-06T10:02:35.988410Z","submitted_at":"2020-10-08T12:46:57Z","title":"Prioritized Level Replay","version":4},"cited_work":{"arxiv_id":"2010.03934","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2010.03934","snapshot_observed_at":"2026-07-04T14:09:52.695084Z","title":"Prioritized level replay","venue":null,"work_id":"791deccd-53da-4462-994c-5b5ae795d4ae","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2010.03934","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2b328687c0252f5a2415f6b3155aea57deb77ac0b168d05549a574ef9ffd1406","observation_id":"e13af358-449a-4dc1-b110-c6013ee8c5fb","resolution":{"observed_at":"2026-05-23T02:42:26.065432Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.03381","last_updated":"2018-12-08T20:16:16Z","snapshot_observed_at":"2026-07-06T07:19:59.972802Z","submitted_at":"2018-12-08T20:16:16Z","title":"Learning Montezuma's Revenge from a Single Demonstration","version":1},"cited_work":{"arxiv_id":"1812.03381","doi":"10.48550/arxiv.1812.03381","metadata_source":"pith","pith_arxiv_id":"1812.03381","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning Montezuma's Revenge from a Single Demonstration","venue":"cs.LG","work_id":"678d4efb-de3e-44d4-b506-208158d68e08","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1812.03381","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:314465966cb17b2a1c46cccf569fe8b0728126a8cf159bde9c7c7f1b06f873b0","observation_id":"d59497fd-ba5c-400e-83c3-3288b371159f","resolution":{"observed_at":"2026-05-23T02:42:26.057246Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13228","last_updated":"2024-07-03T13:46:33Z","snapshot_observed_at":"2026-07-31T05:05:41.080329Z","submitted_at":"2024-02-20T18:42:34Z","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","version":2},"cited_work":{"arxiv_id":"2402.13228","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.13228","snapshot_observed_at":"2026-07-03T16:48:40.350225Z","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","venue":"cs.CL","work_id":"b220de8b-d5cf-4eef-9841-1428c753012c","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2402.13228","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:d9cece357533701722898468961d9d4b45826ef483708c1f07774fbc8117b1e2","observation_id":"b9767396-8989-4e33-86e3-0366a0ea8681","resolution":{"observed_at":"2026-05-23T02:42:26.061849Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:05cea6b267c4c2fe65fa9baae8107e780ee9a3a34e2e58ef5c05017c1302ced2","observation_id":"5ba38af3-159d-467c-9c25-3b7f99c5bbf9","resolution":{"observed_at":"2026-05-23T02:42:26.203304Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12599","last_updated":"2025-06-03T02:14:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T02:48:14Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","version":4},"cited_work":{"arxiv_id":"2501.12599","doi":"10.48550/arxiv.2501.12599","metadata_source":"pith","pith_arxiv_id":"2501.12599","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","venue":"cs.AI","work_id":"bff96ab1-bd6a-4585-be23-74fdb51969c7","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2501.12599","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:d11878b16dc5ad377a74d1934973f74ab538b3b84e443e3297fc44f887d368db","observation_id":"a248f80a-a5e1-4951-93b9-13e1d36b7984","resolution":{"observed_at":"2026-05-23T02:42:26.188553Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:38.585868+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:38.585868+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13818","last_updated":"2026-04-22T00:26:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-18T17:49:55Z","title":"Not All Rollouts are Useful: Down-Sampling Rollouts in LLM Reinforcement Learning","version":5},"cited_work":{"arxiv_id":"2504.13818","doi":"10.48550/arxiv.2504.13818","metadata_source":"pith","pith_arxiv_id":"2504.13818","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Not All Rollouts are Useful: Down-Sampling Rollouts in LLM Reinforcement Learning","venue":"cs.LG","work_id":"e6d53e5b-2180-482b-82ca-0e64d572c87f","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2504.13818","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:335f21c4ec0c8904a9b16a513491f02b8321154470dbb8c6c5fa0c0f4fbba58a","observation_id":"b7293fb5-3be6-4a7c-9258-bd6653cd1ae0","resolution":{"observed_at":"2026-05-23T02:42:26.183846Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The llama 4 herd: The beginning of a new era of natively multimodal ai innovation","venue":null,"work_id":"fdf44679-84b9-4316-85e9-ec5b4c2d27f5","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:f1c678e4a32e4c27b264b08e728fd482fb11ebde6850badea800f2f79e0168da","observation_id":"fccbdeb1-b453-4849-a485-80f6c0eee41e","resolution":{"observed_at":"2026-05-23T02:47:27.511125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.02096","last_updated":"2021-02-04T03:01:31Z","snapshot_observed_at":"2026-07-06T10:20:26.819889Z","submitted_at":"2020-12-03T17:37:01Z","title":"Emergent Complexity and Zero-shot Transfer via Unsupervised Environment Design","version":2},"cited_work":{"arxiv_id":"2012.02096","doi":"10.48550/arxiv.2012.02096","metadata_source":"arxiv_reference","pith_arxiv_id":"2012.02096","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emergent complexity and zero-shot transfer via unsupervised environment design","venue":"arXiv (Cornell University)","work_id":"13fa0948-1511-420f-912e-574ff5f55f77","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2012.02096","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:1d5eda725a446c76d699ec1b0f7535a8e358c87fbfa2afa62d3d8e51054c64d4","observation_id":"017d8ccd-e4a3-4d4e-a1d6-29eda5c39b62","resolution":{"observed_at":"2026-05-23T02:42:26.198274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minigrid &amp; miniworld: Modular &amp; customizable reinforcement learning environments for goal-oriented tasks","venue":null,"work_id":"4c354294-1888-46f6-ab37-b86eec56e0ae","year":2023},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:db66c287e42d1fda62ed5ce2df304286854f9c660b7f77b3093c1e9d18af0b7f","observation_id":"64e71307-ec78-4c6e-9ae0-e2a5656ffd7d","resolution":{"observed_at":"2026-05-23T02:47:27.567335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12044","last_updated":"2024-11-19T09:52:55Z","snapshot_observed_at":"2026-08-04T22:51:31.256089Z","submitted_at":"2023-12-19T10:57:12Z","title":"XLand-MiniGrid: Scalable Meta-Reinforcement Learning Environments in JAX","version":4},"cited_work":{"arxiv_id":"2312.12044","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.12044","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xland-minigrid: Scalable meta-reinforcement learning environments in jax","venue":null,"work_id":"e0014edd-096c-4f70-b56f-63147a2459f0","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2312.12044","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:14d6773e9d0ec70cbe202f8b3b2ed544114999c909ad80b565c596374bda696c","observation_id":"2dbf633d-26d8-46a5-8cd3-57a13e2ed0ed","resolution":{"observed_at":"2026-05-23T02:42:26.048877Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10090","last_updated":"2026-07-04T05:19:25Z","snapshot_observed_at":"2026-07-09T23:17:08.002707Z","submitted_at":"2023-11-16T18:58:43Z","title":"JaxMARL: Multi-Agent RL Environments and Algorithms in JAX","version":6},"cited_work":{"arxiv_id":"2311.10090","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.10090","snapshot_observed_at":"2026-07-07T02:17:02.673620Z","title":"Jaxmarl: Multi-agent rl environments and algorithms in jax","venue":null,"work_id":"2a3824b2-3a1e-4362-be0b-b58a12ec06d0","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2311.10090","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6c82e2991dec8afcea941d19e9cef994603bbb28356e074073a7aff84af98bc3","observation_id":"8d6e0c89-2840-4822-a6f3-7c0dbb8d85cb","resolution":{"observed_at":"2026-07-07T02:17:02.673620Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.01540","last_updated":"2016-06-05T17:54:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-06-05T17:54:48Z","title":"OpenAI Gym","version":1},"cited_work":{"arxiv_id":"1606.01540","doi":"10.1109/jssc.2019","metadata_source":"pith","pith_arxiv_id":"1606.01540","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"OpenAI Gym","venue":"cs.LG","work_id":"6af98f3f-f074-41ae-a689-7dd7b4b8efde","year":2016},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1606.01540","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:98935db090aafa8f6e8695a655b349568f3da14c5e4d91d18f613d61658b4894","observation_id":"66e42759-9de4-40d8-8858-fdd92eea52f3","resolution":{"observed_at":"2026-05-23T02:42:26.173933Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"JAX: composable transformations of Python+NumPy programs","venue":null,"work_id":"a6623f60-5524-4152-889e-6618862f34e6","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:5324cd03d19de7f3a5e7ef2851f5ade06e3c3cc514cc37eb37d4a23f9876e55d","observation_id":"6a88b35b-2f4c-4967-ab08-229c79b51f0a","resolution":{"observed_at":"2026-05-23T02:47:27.507646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04368","last_updated":"2024-11-07T01:58:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T01:58:42Z","title":"Measuring short-form factuality in large language models","version":1},"cited_work":{"arxiv_id":"2411.04368","doi":"10.48550/arxiv.2411.04368","metadata_source":"pith","pith_arxiv_id":"2411.04368","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring short-form factuality in large language models","venue":"cs.CL","work_id":"f8e490ab-7057-43fb-8c6d-06fc603836c7","year":2024},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2411.04368","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:602df14e5ac67b16d59b195380c0a9deef56dc968bc5f130951631a928163825","observation_id":"75b9796e-604a-43ed-a377-36da011eea82","resolution":{"observed_at":"2026-05-23T02:42:26.155841Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.01302","last_updated":"2023-09-30T18:36:42Z","snapshot_observed_at":"2026-07-06T12:43:24.521815Z","submitted_at":"2022-03-02T18:40:00Z","title":"Evolving Curricula with Regret-Based Environment Design","version":3},"cited_work":{"arxiv_id":"2203.01302","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2203.01302","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evolving curricula with regret-based environment design","venue":null,"work_id":"d7dfc56e-54c0-4912-a67d-aabdb2ad9a2d","year":2023},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2203.01302","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:7ff75d655fd5e1734f804f5d3e79901c01c751887d0348aea13fcdf61078ea96","observation_id":"b3d21010-bb44-459f-a500-7b07dff31324","resolution":{"observed_at":"2026-05-23T02:42:26.165070Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03335","last_updated":"2025-10-16T08:23:36Z","snapshot_observed_at":"2026-07-06T21:19:34.329442Z","submitted_at":"2025-05-06T09:08:00Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","version":3},"cited_work":{"arxiv_id":"2505.03335","doi":"10.48550/arxiv.2505.03335","metadata_source":"pith","pith_arxiv_id":"2505.03335","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","venue":"cs.LG","work_id":"b59092c4-76ed-4c78-9006-312bde2e40a6","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2505.03335","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:db420aa02bf8bc8905635a98fd95407460af2428268ab889e32017155784ed3e","observation_id":"e35f324e-bd90-4fd6-bc13-56172bca0ec5","resolution":{"observed_at":"2026-05-23T02:42:26.077279Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:52:59.097203+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:52:59.097203+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9a5767f7-f5b6-4ac1-84e1-8d42bb7939d1","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:f01ad3f4e23d6a8cc02c821e10200046b437032aa0e3f431995cdec0b46e0226","observation_id":"ab6cbce9-9fc2-40f7-97e6-c4081d05ce6f","resolution":{"observed_at":"2026-05-23T02:47:27.563811Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Curriculum learning","venue":null,"work_id":"c5d50b44-2deb-46d3-9178-bba5a7b5d999","year":2009},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:0afebf8b2d4414aafedebd9df59e0798fbe05e018526474ff4c43b8aa3feb55a","observation_id":"3de0a68b-6434-47f1-a568-07ee7e0a358c","resolution":{"observed_at":"2026-05-23T02:47:27.503948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning and development in neural networks: The importance of starting small","venue":null,"work_id":"b42009bf-7d75-4bbc-8142-afd8777578c1","year":1993},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:1ba96a370dacf906b7135ecf43344c02605396fd46dd287ded65b6d6e59a8018","observation_id":"e4019767-2d6d-40b9-84de-c43f166aca00","resolution":{"observed_at":"2026-05-23T02:47:27.500421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Online batch selection for faster training of neural networks","venue":null,"work_id":"bdc8024c-826c-4e23-8abc-f7879656c8c4","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:82a5c3850b49591c472a625fbf063bd8f6ce74b0130cbc97814c118f1872058a","observation_id":"0d215df8-c324-44a5-adaa-71f935de9a1d","resolution":{"observed_at":"2026-05-23T02:47:27.496724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.06343","last_updated":"2016-04-25T14:00:21Z","snapshot_observed_at":"2026-07-06T04:37:08.670695Z","submitted_at":"2015-11-19T20:24:09Z","title":"Online Batch Selection for Faster Training of Neural Networks","version":4},"cited_work":{"arxiv_id":"1511.06343","doi":null,"metadata_source":"pith","pith_arxiv_id":"1511.06343","snapshot_observed_at":"2026-07-09T21:36:34.306361Z","title":"Online Batch Selection for Faster Training of Neural Networks","venue":"cs.LG","work_id":"35a8790d-539d-4362-b36d-6c27d43a6860","year":2015},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1511.06343","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:4d271c8b20e74f3d662517238b8b0d6c767c1c9989c21ceaca887e6dbe41adb4","observation_id":"bda077f8-05ae-47ac-951f-036e3f34c4e6","resolution":{"observed_at":"2026-05-23T02:42:26.116137Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.04371","last_updated":"2020-02-01T21:34:16Z","snapshot_observed_at":"2026-08-02T15:07:57.028674Z","submitted_at":"2019-07-09T19:09:51Z","title":"Ordered SGD: A New Stochastic Optimization Framework for Empirical Risk Minimization","version":5},"cited_work":{"arxiv_id":"1907.04371","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1907.04371","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ordered sgd: A new stochastic optimization framework for empirical risk minimization","venue":null,"work_id":"0a7faa3f-bb32-4da0-b8aa-37b089435795","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1907.04371","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:b9a5bd8174dcbb977c0eddff113fa29d8f469074800487e02aeff15d1140ddb6","observation_id":"56f0f289-b948-4ea1-90a1-8512750e7c1d","resolution":{"observed_at":"2026-05-23T02:42:26.193403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.00762","last_updated":"2019-10-02T03:34:29Z","snapshot_observed_at":"2026-07-06T08:26:17.265696Z","submitted_at":"2019-10-02T03:34:29Z","title":"Accelerating Deep Learning by Focusing on the Biggest Losers","version":1},"cited_work":{"arxiv_id":"1910.00762","doi":null,"metadata_source":"pith","pith_arxiv_id":"1910.00762","snapshot_observed_at":"2026-07-09T21:36:34.301049Z","title":"Accelerating deep learning by focusing on the biggest losers","venue":"cs.LG","work_id":"d74905ea-58c3-4017-8f2d-878a41346b29","year":2019},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1910.00762","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:d4c95028057ac8e494efb1ab364dd0e08bb8c097d1916f34d1b086cfff03ed15","observation_id":"658cc2a6-ce97-4eb2-93cf-fe04e7e4dc14","resolution":{"observed_at":"2026-05-23T02:42:26.150995Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1802.03796","last_updated":"2018-06-08T18:04:50Z","snapshot_observed_at":"2026-07-06T06:22:47.916544Z","submitted_at":"2018-02-11T19:24:47Z","title":"Curriculum Learning by Transfer Learning: Theory and Experiments with Deep Networks","version":4},"cited_work":{"arxiv_id":"1802.03796","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1802.03796","snapshot_observed_at":"2026-07-01T09:45:40.136469Z","title":"Curriculum learning by transfer learning: Theory and experiments with deep networks","venue":null,"work_id":"7626e1e3-853f-4769-b943-dc386d0da54a","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1802.03796","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6ca5c77ff66ca231766f319764d1948d2f27644bb67361a87462bfc17aaa54b2","observation_id":"8390b0f2-a71e-42e1-9618-87e0c85d49cc","resolution":{"observed_at":"2026-05-23T02:42:26.169796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Active learning literature survey","venue":null,"work_id":"afc51652-f571-4d67-ac8a-52526dc22b5a","year":2009},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:6333f27ad5d15a739ee95e437fff23450b35f09ecdd24a4eabf5014c484bb83c","observation_id":"adf83da6-8a75-4fd5-8bb9-4405c3e7615a","resolution":{"observed_at":"2026-05-23T02:47:27.492887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Confidence-based active learning","venue":null,"work_id":"4ef1d214-f0d2-4f36-b6fc-99b9a2cfa557","year":2006},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:9f57b35244a4880ad203d598509df27970ca29366c885051822679cc0614d5b0","observation_id":"afc46c93-d264-4c35-9265-de39385a70cf","resolution":{"observed_at":"2026-05-23T02:47:27.489016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1906.11829","last_updated":"2020-10-27T00:52:20Z","snapshot_observed_at":"2026-07-06T08:03:24.054624Z","submitted_at":"2019-06-26T23:01:47Z","title":"Selection via Proxy: Efficient Data Selection for Deep Learning","version":4},"cited_work":{"arxiv_id":"1906.11829","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1906.11829","snapshot_observed_at":"2026-07-03T19:28:52.634253Z","title":"Selection via proxy: Efficient data se- lection for deep learning.arXiv preprint arXiv:1906.11829","venue":null,"work_id":"12daad0e-7402-46c0-bc01-a0fd7dfbadf7","year":1906},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1906.11829","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a64484e44b04234bbc5a1de8b7b3593083ea2fb8c731e0b29577cc9d1dae6469","observation_id":"0cd69aea-9ee3-423a-9512-d3105fcdcb0a","resolution":{"observed_at":"2026-05-23T02:42:26.121575Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07137","last_updated":"2022-09-26T17:28:16Z","snapshot_observed_at":"2026-07-06T13:20:51.362304Z","submitted_at":"2022-06-14T19:49:52Z","title":"Prioritized Training on Points that are Learnable, Worth Learning, and Not Yet Learnt","version":3},"cited_work":{"arxiv_id":"2206.07137","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2206.07137","snapshot_observed_at":"2026-06-30T22:05:05.665419Z","title":"Prioritized training on points that are learnable, worth learning, and not yet learnt","venue":null,"work_id":"d84eead5-4940-4998-8d33-0d0bdb183105","year":2022},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2206.07137","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:69267474ea67f09d7edd16dd6c2a5326fad586d7b5c757137095ec5c5b176899","observation_id":"b9113709-88e6-462b-9b33-c67eb55989bb","resolution":{"observed_at":"2026-05-23T02:42:26.126116Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.04759","last_updated":"2019-05-14T16:34:15Z","snapshot_observed_at":"2026-07-06T06:55:37.123626Z","submitted_at":"2018-08-14T15:45:48Z","title":"An Overview and a Benchmark of Active Learning for Outlier Detection with One-Class Classifiers","version":2},"cited_work":{"arxiv_id":"1808.04759","doi":null,"metadata_source":"pith","pith_arxiv_id":"1808.04759","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An Overview and a Benchmark of Active Learning for Outlier Detection with One-Class Classifiers","venue":"cs.LG","work_id":"4c3800e8-6646-418b-9b46-2274f7cbd007","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1808.04759","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2b0413f891ba2a94e01da3652e52a88a9bf3377ea40ec7b09fb9618b28bcf88d","observation_id":"898ef982-bd6f-4e4f-b954-3510565f7a36","resolution":{"observed_at":"2026-05-23T02:42:26.111402Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Training deep models faster with robust, approximate importance sampling","venue":null,"work_id":"421de673-3a9d-428d-9f1a-8675cee38ad3","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:de9fa671ab0d34bb4cc3345c119bf961b1a78234c23f159da4ce1f4af549a5bd","observation_id":"f064bdf0-fd28-4c58-9430-a0768496b8c0","resolution":{"observed_at":"2026-05-23T02:47:27.622517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Not all samples are created equal: Deep learning with importance sampling","venue":null,"work_id":"77a9584e-eddb-4d2a-8996-f679cae25ec5","year":2018},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:93b5576d3a1f8eeb0599be839b6fe0907c0b303d0972bdcf2b022b525ea0cd61","observation_id":"2a4d60cb-1c34-478e-bd5e-6b3a0e26c75d","resolution":{"observed_at":"2026-05-23T02:47:27.618558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-paced learning for latent variable models","venue":null,"work_id":"a6496fab-4b6f-4710-98d4-274e7045732e","year":2010},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:b99347766304742fa8cd75e3bcdb25e7f3cec6c9ddfdcb778558a13e43d8ef49","observation_id":"341adf48-c1a3-4997-815f-3196b743385b","resolution":{"observed_at":"2026-05-23T02:47:27.634546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1704.03003","last_updated":"2017-04-10T18:25:29Z","snapshot_observed_at":"2026-08-01T21:11:47.069345Z","submitted_at":"2017-04-10T18:25:29Z","title":"Automated Curriculum Learning for Neural Networks","version":1},"cited_work":{"arxiv_id":"1704.03003","doi":null,"metadata_source":"pith","pith_arxiv_id":"1704.03003","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Automated Curriculum Learning for Neural Networks","venue":"cs.NE","work_id":"3e4c1efa-45d7-4f02-8a5a-14a47714e09e","year":2017},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1704.03003","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:1e52d0adb93835db0289f4153b76d91a9c9475f54190da78aedb92d700ab7e5f","observation_id":"ad35adb3-e42d-4c0e-92c1-9c32b935e279","resolution":{"observed_at":"2026-05-23T02:42:26.130839Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.00183","last_updated":"2017-11-29T20:57:09Z","snapshot_observed_at":"2026-07-06T05:49:21.582343Z","submitted_at":"2017-07-01T18:13:17Z","title":"Teacher-Student Curriculum Learning","version":2},"cited_work":{"arxiv_id":"1707.00183","doi":null,"metadata_source":"pith","pith_arxiv_id":"1707.00183","snapshot_observed_at":"2026-06-28T23:52:49.216887Z","title":"Teacher-Student Curriculum Learning","venue":"cs.LG","work_id":"7c8c807e-3c64-4652-92a5-0bfcbb669352","year":2017},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1707.00183","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:cf79a34ff61445158eacb36210d561a11575eaf9ea0e58b30735c9445a35dbc7","observation_id":"e480ceff-129f-4204-8390-55f1b6bf2d4f","resolution":{"observed_at":"2026-05-23T02:42:26.097838Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A survey of multi-task deep reinforcement learning","venue":null,"work_id":"3e21347e-ba41-414f-a783-c5d8e94e246b","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:287d85cedf01ddd107ace94f594f615f940631f05c1b6389ef5c36bacaa79154","observation_id":"f7a01818-4359-47b6-a735-ded8010df04b","resolution":{"observed_at":"2026-05-23T02:47:27.611853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Automatic curriculum learning through value disagreement","venue":null,"work_id":"a783c50f-5b1d-4e9a-a23f-9a327c73498d","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2a8fa09a04bc127b67343e48d7a6cf2cb27d2c1e87ca971a1099f910e56ce5f1","observation_id":"4d81e9e8-2b97-4458-813d-3fd6076f4007","resolution":{"observed_at":"2026-05-23T02:47:27.604553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.09641","last_updated":"2020-06-17T03:58:25Z","snapshot_observed_at":"2026-07-06T09:29:54.899176Z","submitted_at":"2020-06-17T03:58:25Z","title":"Automatic Curriculum Learning through Value Disagreement","version":1},"cited_work":{"arxiv_id":"2006.09641","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2006.09641","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Available: https://arxiv.org/abs/2006.09641","venue":null,"work_id":"a15a584e-1aca-4b23-af6f-236059b7fb3a","year":2006},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2006.09641","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:c57b9268978078059b52ee8661a5ecbedcd0a9c029f47a4f3aa20093944685e1","observation_id":"c3e9b355-ff56-4a18-8a1b-b0961559ae68","resolution":{"observed_at":"2026-05-23T02:42:26.106419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2011.01054","last_updated":"2021-07-01T13:45:56Z","snapshot_observed_at":"2026-07-06T10:10:56.466018Z","submitted_at":"2020-11-02T15:37:37Z","title":"Information-theoretic Task Selection for Meta-Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2011.01054","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2011.01054","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Information-theoretic task selection for meta-reinforcement learning","venue":null,"work_id":"f121ce98-6373-4a6f-b5aa-cdd744372e22","year":2021},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2011.01054","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a5b8c158c8fe1710229ed6265ddfdba5ccbb62ffd112e628291ff8f789d3688c","observation_id":"64a38f54-5c17-4084-a444-728db0e96793","resolution":{"observed_at":"2026-05-23T02:42:26.093984Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.02832","last_updated":"2020-07-06T15:36:05Z","snapshot_observed_at":"2026-08-03T21:30:52.105566Z","submitted_at":"2020-07-06T15:36:05Z","title":"Maximum Entropy Gain Exploration for Long Horizon Multi-goal Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2007.02832","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.02832","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Maximum entropy gain exploration for long horizon multi-goal reinforcement learning","venue":null,"work_id":"dd3f8939-373d-452c-b019-2651fe090a29","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2007.02832","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:045c7db6137ea519e9bcb23b7a7a598e6d179b34b97829f6088ba8005fe2be9c","observation_id":"70dea3b6-cd80-46cf-8035-3d6f2862fc7a","resolution":{"observed_at":"2026-05-23T02:42:26.135156Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1903.03698","last_updated":"2020-08-04T04:07:27Z","snapshot_observed_at":"2026-07-06T07:38:07.496179Z","submitted_at":"2019-03-08T23:32:17Z","title":"Skew-Fit: State-Covering Self-Supervised Reinforcement Learning","version":4},"cited_work":{"arxiv_id":"1903.03698","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1903.03698","snapshot_observed_at":"2026-07-04T06:59:38.343947Z","title":"arXiv preprint arXiv:1903.03698 , year=","venue":null,"work_id":"9912f351-bd16-4037-9627-98a975a611ff","year":1903},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1903.03698","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:d0c9181c0386e1d00b5792b3acbdb568437d0c9e1b5f04aaea2a374562229617","observation_id":"1f05f239-4449-48a3-b8f2-60e199228dce","resolution":{"observed_at":"2026-05-23T02:42:26.146328Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1901.09720","last_updated":"2019-03-25T10:04:05Z","snapshot_observed_at":"2026-07-06T07:29:28.340655Z","submitted_at":"2019-01-28T15:00:29Z","title":"CLIC: Curriculum Learning and Imitation for object Control in non-rewarding environments","version":4},"cited_work":{"arxiv_id":"1901.09720","doi":null,"metadata_source":"pith","pith_arxiv_id":"1901.09720","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"CLIC: Curriculum Learning and Imitation for object Control in non-rewarding environments","venue":"cs.LG","work_id":"492ecd38-47d2-4b49-8cee-c2b63a38e7e0","year":2019},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1901.09720","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:5eb9e0241b474e091a4ae422163b5dfa6c83b5ec6004573b039dfc8d53962919","observation_id":"727e61de-24b1-4793-b80a-fdf8899f22fe","resolution":{"observed_at":"2026-05-23T02:42:26.101950Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.01114","last_updated":"2020-10-02T17:17:45Z","snapshot_observed_at":"2026-07-06T10:00:59.593985Z","submitted_at":"2020-10-02T17:17:45Z","title":"Goal-GAN: Multimodal Trajectory Prediction Based on Goal Position Estimation","version":1},"cited_work":{"arxiv_id":"2010.01114","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2010.01114","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Goal-gan: Multimodal trajectory prediction based on goal position estimation","venue":null,"work_id":"e8fe4b86-e2b9-48dd-9bf1-7294ea813fd4","year":2020},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/2010.01114","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a335e74a52e844a6dfc95fe09da60a578670e537a7be2a326729339ad3087a2d","observation_id":"c34e21d6-43e9-4e82-bc52-0fda7635b0e5","resolution":{"observed_at":"2026-05-23T02:42:26.034216Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1511.05952","last_updated":"2016-02-25T17:55:31Z","snapshot_observed_at":"2026-07-06T04:37:00.543211Z","submitted_at":"2015-11-18T20:54:44Z","title":"Prioritized Experience Replay","version":4},"cited_work":{"arxiv_id":"1511.05952","doi":"10.48550/arxiv.1511.05952","metadata_source":"pith","pith_arxiv_id":"1511.05952","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Prioritized Experience Replay","venue":"cs.LG","work_id":"927187c1-c50e-4ca7-b0fa-55589957731f","year":2015},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"cited_paper":"/paper/1511.05952","citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:8db759e83c580b2a0579f0434335ba3f60b36f7fd1e51cebc46889b1d1536394","observation_id":"7124d511-8153-4b2c-9e55-ffa1792fb33e","resolution":{"observed_at":"2026-05-23T02:42:26.041639Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In [48] the authors use the loss from a pre-trained model to estimate the difficulty of new samples for a freshly initialized network learning a new task","venue":null,"work_id":"3084f4f9-5cd6-44d3-bd25-274e2c33e000","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:9af96c4c63764472e68fa330b9804e47c8693a21719951c827a44fb3b5a56e99","observation_id":"cf3a376b-7f5e-449c-bb41-3270b1bba14a","resolution":{"observed_at":"2026-05-23T02:47:27.597336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LILO can be seen as using return variance—or learnability—as an estimator of entropy or uncertainty","venue":null,"work_id":"88f7773e-43c6-46f8-be64-42a932336ffc","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:25455ccb0cb1d1aa11d63c605d25086a8e31d205458b44b2169d9013d6545303","observation_id":"e4942561-ca24-4a84-8d16-c660bc6881de","resolution":{"observed_at":"2026-05-23T02:47:27.580531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This allows prioritizing samples that maximize the change in loss—i.e., the model’s learning progress [ 52, 11, 53, 49]","venue":null,"work_id":"ddde022e-5e71-4662-900c-9654e5dffcb0","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:da5f149af3ca9fb4085de3b2f1144445f3fcd1ab6f12f5a81a7ca8a9260491d6","observation_id":"e6b2d2c2-ed7d-4ac1-ad38-a25ab2db34a8","resolution":{"observed_at":"2026-05-23T02:47:27.585243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-paced learning [56] is an early approach that allows the model to determine the pace at which it incorporates harder examples with higher values of U","venue":null,"work_id":"29f7bfda-d74b-484f-b502-d6f1afd3ac67","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:8a190974148a5cbccdb79f0a946ae5c5eabb7db1d31cb2a6f4565b0bd9953b60","observation_id":"249db8c9-8b4f-4918-ba51-3441497bc233","resolution":{"observed_at":"2026-05-23T02:47:27.608148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Each bullet point contains a claim and a hyperlink to the section of the paper that proves the claim","venue":null,"work_id":"95be00f2-83cb-4e92-8c1b-962361088c8b","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:5fb37f6d7445f792304a6af8db757355b1623511220df93ff5e5c6b08eb761e1","observation_id":"c813b16f-bd7c-49aa-a48c-5d1e9b73ad32","resolution":{"observed_at":"2026-05-23T02:47:27.534779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Section 7 also contains some limitations","venue":null,"work_id":"72919680-a787-4129-a892-d323813cc5dd","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:8952151706eeab899dd5e3722a9ed651023b7d4fc209814b8b1de4360ea07e25","observation_id":"5eb58b2e-1dfe-408b-a88e-b0a9712c4ceb","resolution":{"observed_at":"2026-05-23T02:47:27.527386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper does not include theoretical results","venue":null,"work_id":"6f5ecb1d-54fe-4281-bdbb-8352a3ad2d19","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:80c416d1317b8550bae15d5ca0c8e2c5ab23a0e281fddd2656cd77540daf64ef","observation_id":"78f48f7a-df59-48fa-9422-db9011825e5b","resolution":{"observed_at":"2026-05-23T02:47:27.542651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The results in 6 were produced using open-source codebases [5] [4] and models, with some small additions of code by us","venue":null,"work_id":"f1bb5ce0-895b-4156-b239-b8a2c3bcdbdf","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2421f1371fac2c67b05dcbb8e8476ebf0ca92927bf8ca717d59c34d3b4a96934","observation_id":"3145bef9-d59f-475a-9721-c673f544bdd3","resolution":{"observed_at":"2026-05-23T02:47:27.514929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This is described in Section 5","venue":null,"work_id":"20731e47-00c6-4fb2-93e2-764a40cea446","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:367bb7bed31a9e4b9c21209b3269c3ac7d57fc9b1d6dcdefed7359520ad94b45","observation_id":"a8ba68fb-4e0a-47d1-823d-1ead5b3a6ea7","resolution":{"observed_at":"2026-05-23T02:47:27.546614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"All the other hyperparameters for training are replicated directly from the VinePPO [ 5] and Oat [4] libraries, and the user is directed to these in Section 5","venue":null,"work_id":"0dc7df98-2c3b-4a1f-b79e-6643b87ad5ca","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:4b3be91ad98d1023187ae27d62fb7715a2516c85e0f3b7f0aae3eab7cf2fe187","observation_id":"dc7ecaf9-0b14-44d1-9a44-9487efbe03b1","resolution":{"observed_at":"2026-05-23T02:47:27.560504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"We have, however, provided training curves to aid the reader in interpreting the significance of the results","venue":null,"work_id":"844637ff-2056-418f-b638-1760d879b162","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:9e7909c647771d83a4ca9362ead6770c2bf03f9f6dcecf66b743837f50f9d206","observation_id":"d9bdbbaf-dec8-49ca-87ef-49ab1f1754ed","resolution":{"observed_at":"2026-05-23T02:47:27.614972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"• The paper should indicate the type of compute workers CPU or GPU, internal cluster, or cloud provider, including relevant memory and storage","venue":null,"work_id":"e8418996-d16a-4999-af18-2c4176d948f2","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:513cb8c46823e28399fa60805cae2597304c60b57a1af9b035926fb7b6e6a448","observation_id":"d8e05a43-969a-4c9a-bbe6-3b08774999e8","resolution":{"observed_at":"2026-05-23T02:47:27.626385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the authors have not reviewed the NeurIPS Code of Ethics","venue":null,"work_id":"fc323e45-6aa1-42d5-94cc-27d04f7b239c","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:c0cc776bcecb2703c67a5675b38eeb79d315b8fe3e019eaaae33f91764c845e0","observation_id":"4937820f-e32d-4279-bc46-91c07d4beddf","resolution":{"observed_at":"2026-05-23T02:47:27.538684Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that there is no societal impact of the work performed","venue":null,"work_id":"3a5f2b2a-8a39-4c2c-93c8-1dd551023fe5","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:bb09ec97e277ef3bdc72180b5d7bc06727544bd93b85d52d355f1d64132a81f4","observation_id":"a8a36f1f-ba68-4b49-a227-65ae6d3846c6","resolution":{"observed_at":"2026-05-23T02:47:27.589234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper poses no such risks","venue":null,"work_id":"16d0842b-1f0d-4877-874c-f1277e7dee19","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:71d5132f85893ac64a6cdc367476233f8c7115c89b25bd648abb093095b5ca9b","observation_id":"7e20c96e-748b-44cd-8914-0ed7640fc2a7","resolution":{"observed_at":"2026-05-23T02:47:27.593649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The two libraries we used for training (VinePPO and Oat) are both fully open-source","venue":null,"work_id":"31ea4de4-36cc-4efb-8f2c-51ba27c987fa","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:2caf971c38adcec684a3699839158537efa92529ef80ff6222ea1295c7a585d7","observation_id":"6f8bbaa6-8709-40d1-8dab-d2b6437daf55","resolution":{"observed_at":"2026-05-23T02:47:27.570800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"It very simple, and could be implemented from this paper alone","venue":null,"work_id":"7f5083ca-2951-4497-ab0c-ebfb6a8ebe26","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:331c096ef3bdacac23578ccd3573e1bbc888008e33fb5409723e29049dc42aab","observation_id":"990cf32f-c5b3-4c03-90c2-0d0db59eac4a","resolution":{"observed_at":"2026-05-23T02:47:27.576905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper does not involve crowdsourcing nor research with human subjects","venue":null,"work_id":"88de757b-5c29-477c-b32b-e15a66aaf1a2","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a86b3c22fea33a66f47b7dfee6eb2deb7a720af0b1cefb53197237b14afdf5ee","observation_id":"25126526-67c4-4dd8-9def-b2d074d6cd84","resolution":{"observed_at":"2026-05-23T02:47:27.600904Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guidelines: • The answer NA means that the paper does not involve crowdsourcing nor research with human subjects","venue":null,"work_id":"aaefd557-315d-4a57-bbfe-c97a15ef0695","year":null},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:a63dcc48e61536f7973a8e1783f044cc983c09f95c3a247d9aa80ae8b2024ef3","observation_id":"8aa1cc51-a145-43a9-813f-4d0b0bb8c35d","resolution":{"observed_at":"2026-05-23T02:47:27.630722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Answer: [NA] Justification: LLM usage was only used in a standard way for editing","venue":null,"work_id":"06999c0b-724b-4e93-92f5-b974287a8c26","year":2025},"citing_paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability","version":6},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-23T02:41:21.571824Z"},"links":{"citing_paper":"/paper/2502.12272"},"observation_digest":"sha256:c0969279acce93cb97bdaa790b368cb6ab05b096cd40216a640726d03584435b","observation_id":"a35994fa-622d-45e2-b5de-d565cf4ea1f2","resolution":{"observed_at":"2026-05-23T02:47:27.531069Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.12272","last_updated":"2026-04-30T11:57:30Z","latest_version":6,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-17T19:16:37Z","title":"Learning to Reason at the Frontier of Learnability"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":1,"verified_exact":46,"verified_fuzzy":39},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 1 inbound Pith citation observation for arXiv:2502.12272."}