{"as_of":"2026-08-04T13:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:53ce032a275e0a78dbd8ab620ce0f552c2c47635cdef25c76d98628b290723c6","coverage":[{"denominator":46,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":46,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T11:50:26.030339Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.01075/citation-record","integrity":"/paper/2606.01075/integrity","json":"/paper/2606.01075/citation-record.json","paper":"/paper/2606.01075"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.13388","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T23:06:20.548395Z","title":"Miranda, Hanyang Zhao, Mohammad Rifqi Farhansyah, Garry Kuwanto, Derry Wijaya, and Genta Indra Winata","venue":null,"work_id":"5392a515-f4de-4825-aaca-5ddece59ac6e","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:10d1f120990e2127179b28b8f35f22af28249292076d5931c477988c093d2308","observation_id":"9b2654f6-84c6-40e9-af8f-46ba3b22ae61","resolution":{"observed_at":"2026-06-28T17:22:24.933942Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T17:18:37.671369Z","title":"Annotating the annotators: Analysis, insightsandmodellingfromanannotationcampaignonpersuasiontechniques detection","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:a5a5a96c870be4b2cf5e08e806f318dc5dbb8c73bee5a96dc9b867951d03b042","observation_id":"8ef754e8-b454-4eca-96e8-e60de265ad14","resolution":{"observed_at":"2026-06-28T17:18:37.671369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T17:18:37.671369Z","title":"Avrim Blum, Daniel Hsu, Cyrus Rashtchian, and Donya Saless","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:7a880181bde49bd828df8f05293ae1e7595cd4df00c57670da865bf827058a08","observation_id":"79d82125-d43a-4c5f-a8f4-f8e9ea9cc3dc","resolution":{"observed_at":"2026-06-28T17:18:37.671369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:340a47b14c4cc80b3bf5e35d4e8f6aa3e89071de10c4abf430704cb3881ff863","observation_id":"da42ab51-9ed2-4ed8-bb2c-146d8f3330f0","resolution":{"observed_at":"2026-06-28T17:22:24.862173Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:97fa74e2ad5fa2001f60776bed53cb994cfbe4590d5ca496a8f82803f0b1d01e","observation_id":"ac52c175-ccf9-44d8-8445-67eac2bf696f","resolution":{"observed_at":"2026-06-28T17:22:24.884454Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:f57b2d5ec6bc22e65779dd87d4fea10fbb08fc161d70af884b23ebb89880dbd8","observation_id":"22d9bd67-12de-4616-86ab-0d46057438c9","resolution":{"observed_at":"2026-06-28T17:22:24.870434Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15115","last_updated":"2024-11-27T11:58:50Z","snapshot_observed_at":"2026-07-06T19:36:27.984192Z","submitted_at":"2024-10-19T13:53:50Z","title":"On Designing Effective RL Reward at Training Time for LLM Reasoning","version":3},"cited_work":{"arxiv_id":"2410.15115","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.15115","snapshot_observed_at":"2026-07-04T20:20:07.365992Z","title":"net/forum?id=6Tm1mposlrM","venue":null,"work_id":"5c46cc3b-db37-4655-a0ff-f28ab6c32a1f","year":2024},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2410.15115","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:6f191da882d0bde709bf380e4a122408c9657282c101d7ed3cbbed35cdeef599","observation_id":"39980ace-226b-4322-8de7-cf919e314565","resolution":{"observed_at":"2026-06-28T17:22:24.859815Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19786","last_updated":"2025-03-25T15:52:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-25T15:52:34Z","title":"Gemma 3 Technical Report","version":1},"cited_work":{"arxiv_id":"2503.19786","doi":"10.1007/978-3-540-48085-3_36","metadata_source":"pith","pith_arxiv_id":"2503.19786","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Gemma 3 Technical Report","venue":"cs.CL","work_id":"f93e08bf-9e96-409b-8ac6-b8385fd17fd7","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2503.19786","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:d7910ef46224a69240020afef52de9f05dbdf88052bdde028268d76dfe6926a4","observation_id":"2cb1455c-c6ab-4c98-8adc-ecf9c6831f92","resolution":{"observed_at":"2026-06-28T17:22:24.924563Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04178","last_updated":"2025-06-05T02:21:52Z","snapshot_observed_at":"2026-08-03T02:41:13.331563Z","submitted_at":"2025-06-04T17:25:39Z","title":"OpenThoughts: Data Recipes for Reasoning Models","version":2},"cited_work":{"arxiv_id":"2506.04178","doi":"10.48550/arxiv.2506.04178","metadata_source":"pith","pith_arxiv_id":"2506.04178","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"OpenThoughts: Data Recipes for Reasoning Models","venue":"cs.LG","work_id":"c7acbe41-27a0-4773-a7be-8f08d86cdf21","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2506.04178","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:b6962c7daadbb31df2fcd08b92d49245ae79e1b2a113ba19715d5ef05f73aa5c","observation_id":"b111a94c-994f-4f75-8016-ab24774a1a39","resolution":{"observed_at":"2026-06-28T17:22:24.870793Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T10:52:48.169741+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T10:52:48.169741+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.17746","last_updated":"2025-10-03T01:55:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-23T17:57:55Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","version":2},"cited_work":{"arxiv_id":"2507.17746","doi":"10.48550/arxiv.2507.17746","metadata_source":"pith","pith_arxiv_id":"2507.17746","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","venue":"cs.LG","work_id":"805a846c-dae9-4375-abd8-a86dc6934496","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2507.17746","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:4a4eff87088146d976e9d29f22242f11512024c7001580e74c1420795079d456","observation_id":"536da892-ee22-4e77-a2d9-3ba2cea9676c","resolution":{"observed_at":"2026-06-28T17:22:24.856776Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:02.879284+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:02.879284+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-07-10T16:57:24.565388Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:488284f7bfa9607a7696d5d56b289c2c24304fa6c4820d12a6ee6b691fba3783","observation_id":"9a0c1eb6-0ce2-442e-9a96-8f627c9a2a06","resolution":{"observed_at":"2026-06-28T17:22:24.904800Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01951","last_updated":"2024-12-04T14:20:21Z","snapshot_observed_at":"2026-07-06T20:00:32.622563Z","submitted_at":"2024-12-02T20:24:17Z","title":"Self-Improvement in Language Models: The Sharpening Mechanism","version":2},"cited_work":{"arxiv_id":"2412.01951","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.01951","snapshot_observed_at":"2026-07-03T14:08:21.732193Z","title":"Self-improvement in language models: The sharpening mechanism","venue":null,"work_id":"7f90ec4b-7277-4b8a-b651-135711d76650","year":2024},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2412.01951","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:484bbd453cda23c819ca816d1670be762772bf41508ff9736b936055873ed8ce","observation_id":"73f72c26-913e-4e00-9097-f07e002fca28","resolution":{"observed_at":"2026-06-28T17:22:24.862765Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21878","last_updated":"2025-04-07T17:44:38Z","snapshot_observed_at":"2026-07-30T21:58:54.192349Z","submitted_at":"2025-03-27T18:00:08Z","title":"Is Best-of-N the Best of Them? Coverage, Scaling, and Optimality in Inference-Time Alignment","version":2},"cited_work":{"arxiv_id":"2503.21878","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.21878","snapshot_observed_at":"2026-07-02T05:16:38.985376Z","title":"arXiv preprint arXiv:2503.21878 , year=","venue":null,"work_id":"a609293c-a7fe-4aab-a912-90348355a2f5","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2503.21878","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:a5df7b29e875ef06aa051913b2e3c3de86673e3b0be2d4ebb9274f91d39ffe11","observation_id":"5aa575b6-57f0-4a85-9993-76f83eeccc51","resolution":{"observed_at":"2026-06-28T17:22:24.893889Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.14234","last_updated":"2026-05-09T12:06:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-17T17:59:42Z","title":"Compute as Teacher: Turning Inference Compute Into Reference-Free Supervision","version":3},"cited_work":{"arxiv_id":"2509.14234","doi":"10.18653/v1/2020.coling-main.580.https://","metadata_source":"pith","pith_arxiv_id":"2509.14234","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Compute as Teacher: Turning Inference Compute Into Reference-Free Supervision","venue":"cs.LG","work_id":"1950c695-dc64-40eb-a4ea-acbc4025cf6f","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2509.14234","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:227f5dc375851c7eca234562c8e1bcf85a7ac01e88958b92bd22dc9dd969783a","observation_id":"d418c805-f767-42cb-8fc5-743144cb575d","resolution":{"observed_at":"2026-06-28T17:22:24.904073Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.00606","last_updated":"2024-06-06T22:48:35Z","snapshot_observed_at":"2026-07-06T18:23:52.659446Z","submitted_at":"2024-06-02T03:36:37Z","title":"LLMs Could Autonomously Learn Without External Supervision","version":2},"cited_work":{"arxiv_id":"2406.00606","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.00606","snapshot_observed_at":"2026-06-28T17:22:24.894937Z","title":"Llms could autonomously learn without external supervision.arXiv preprint arXiv:2406.00606,","venue":null,"work_id":"2350e09a-a2e3-494f-9b83-531b6691f5bc","year":null},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2406.00606","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:95d6a0dea1e1909eeec182f7b4681f92d56770ca95cca4813918ecc45f07f1bf","observation_id":"0d110353-efde-4ecc-9ad3-290604fad18d","resolution":{"observed_at":"2026-06-28T17:22:24.896495Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T17:18:37.671369Z","title":"Demystifying synthetic data in llm pre-training: A systematic study of scaling laws, benefits, and pitfalls","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:f072e7052a3355abe0c4c65f179483864345c09e9b686b13795e6f57117bb112","observation_id":"5b2fbd29-b6ea-48d7-b0b2-79804be7b781","resolution":{"observed_at":"2026-06-28T17:18:37.671369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.11824","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T17:22:24.899612Z","title":"Search-based correction of reasoning chains for language models.arXiv preprint arXiv:2505.11824,","venue":null,"work_id":"fcd887ac-4444-4657-9132-70e2f4b08097","year":null},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:5ca60cd943fbc5234a8f01e1773cd6796bb0c8c15a8ab6714c7666f8aa808b3e","observation_id":"9ccf7c5f-5c9c-4925-9c20-6319af8bde5b","resolution":{"observed_at":"2026-06-28T17:22:24.901017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.07414","doi":"10.48550/arxiv.2509.07414","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Language self-play for data-free training","venue":null,"work_id":"9dc2373f-5f64-4169-9c74-76e126641746","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:ce5cb190ae2b1d88ea4a8d34e9aab1b4ab6dbe36f08ff5b34583a26329fb2e1e","observation_id":"2767187f-e6d4-44ec-8358-2a2a6d1a939f","resolution":{"observed_at":"2026-06-28T17:22:24.834128Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:02.109615+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:02.109615+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12917","last_updated":"2024-10-04T17:28:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-19T17:16:21Z","title":"Training Language Models to Self-Correct via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2409.12917","doi":"10.48550/arxiv.2409.12917","metadata_source":"pith","pith_arxiv_id":"2409.12917","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Training Language Models to Self-Correct via Reinforcement Learning","venue":"cs.LG","work_id":"3ac87f3c-6dc4-492a-bbfb-8cdc05a15706","year":2024},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2409.12917","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:a3f4ac7dfbc966ab0464bb5dbfe901f3daed02ceada60cc269c981e53adb7e11","observation_id":"30caf84e-ec89-42ce-b53f-7c8efb986e0a","resolution":{"observed_at":"2026-06-28T17:22:24.915769Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:38.810784+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:38.810784+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-07-06T19:55:37.400185Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":"2411.15124","doi":"10.48550/arxiv.2411.15124","metadata_source":"pith","pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","venue":"cs.CL","work_id":"28c9dbea-056a-48c2-8000-85f809827e45","year":2024},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:9efdb5b73589c599110272fc43bb8d9e254db1ad896f66ac3b15220674b21f7f","observation_id":"ec45cdcf-ec91-4ab9-b5cb-639a6397113f","resolution":{"observed_at":"2026-06-28T17:22:24.909914Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14565","last_updated":"2025-07-15T06:30:11Z","snapshot_observed_at":"2026-07-06T20:39:47.920307Z","submitted_at":"2025-02-20T13:50:02Z","title":"ReVISE: Learning to Refine at Test-Time via Intrinsic Self-Verification","version":2},"cited_work":{"arxiv_id":"2502.14565","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14565","snapshot_observed_at":"2026-07-04T00:19:14.046607Z","title":"ReVISE: Learning to refine at test-time via intrinsic self-verification.arXiv preprint arXiv:2502.14565","venue":null,"work_id":"80b6b3d5-b525-4e88-89bc-38e7653a7a2a","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2502.14565","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:19a00334c58310089433e7e510b78680b7696e40827ef3a22a0807d71a1dcadf","observation_id":"aec1a5f7-acac-4f2c-8072-27ecff59d401","resolution":{"observed_at":"2026-06-28T17:22:24.906664Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20379","last_updated":"2025-02-27T18:53:30Z","snapshot_observed_at":"2026-07-31T02:25:03.070138Z","submitted_at":"2025-02-27T18:53:30Z","title":"Multi-Agent Verification: Scaling Test-Time Compute with Multiple Verifiers","version":1},"cited_work":{"arxiv_id":"2502.20379","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.20379","snapshot_observed_at":"2026-07-02T20:07:21.395861Z","title":"Demystifying long chain-of-thought reasoning in LLMs","venue":null,"work_id":"f7415b1c-bfb6-40bd-b4a5-822525032b31","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2502.20379","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:f849aa04295a17bc8dd55f413b3fceb1e4c65e4d528033dea74b16fdfedcf835","observation_id":"071efbf6-4c99-49ef-9fb6-a0e5c6ca72cb","resolution":{"observed_at":"2026-06-28T17:22:24.899116Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.05145","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T17:22:24.843563Z","title":"Self-improving vlm judges without human annotations.arXiv preprint arXiv:2512.05145,","venue":null,"work_id":"d96e045a-b62c-40c9-b015-db553272767e","year":null},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:7fcc7f6b3b5ff9755386a69b37b77dbdf5042b84f1d80db3c18abef1d0c2e7a0","observation_id":"38a7524c-f242-49d8-bcce-59accbd5e717","resolution":{"observed_at":"2026-06-28T17:22:24.845015Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24864","last_updated":"2025-05-30T17:59:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-30T17:59:01Z","title":"ProRL: Prolonged Reinforcement Learning Expands Reasoning Boundaries in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.24864","doi":"10.48550/arxiv.2505.24864","metadata_source":"pith","pith_arxiv_id":"2505.24864","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"ProRL: Prolonged Reinforcement Learning Expands Reasoning Boundaries in Large Language Models","venue":"cs.CL","work_id":"b6fae4a8-0f64-45dd-b9f4-22f8e0a7e7f6","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2505.24864","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:d7ae4f3c313ba6f56b335bf9725faf2d3211cf4a341d0bf9c08cb9192a6d3b9c","observation_id":"818e8dfd-a375-4be8-b94c-906bad8d0842","resolution":{"observed_at":"2026-06-28T17:22:24.820455Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.14610","last_updated":"2023-03-02T07:41:55Z","snapshot_observed_at":"2026-08-02T10:24:59.821810Z","submitted_at":"2022-09-29T08:01:04Z","title":"Dynamic Prompt Learning via Policy Gradient for Semi-structured Mathematical Reasoning","version":3},"cited_work":{"arxiv_id":"2209.14610","doi":null,"metadata_source":"pith","pith_arxiv_id":"2209.14610","snapshot_observed_at":"2026-07-05T17:41:17.457873Z","title":"arXiv preprint arXiv:2209.14610 , year =","venue":"cs.LG","work_id":"b02aa44e-3fa1-4e2f-a1d4-adc73c214fdc","year":2022},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2209.14610","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:adfae14ec16619fbfd3df75dd0724936dc393f2f474fe72bfa5b5563fbced8e3","observation_id":"03258f5a-a419-4f52-962e-1e01d6164398","resolution":{"observed_at":"2026-06-28T17:22:24.934086Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T17:18:37.671369Z","title":"The “problem” of human label variation: On ground truth in data, modeling and evaluation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:32963fc901e4c451c68cf3184051bae5238728facf1b74bf033c35cfa6d98845","observation_id":"14c942b6-4f89-4d4b-b320-f6db9d609586","resolution":{"observed_at":"2026-06-28T17:18:37.671369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.18203","doi":"10.48550/arxiv.2506.18203","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"The answer is (X)\\","venue":null,"work_id":"a91fdb82-8449-4f60-be1a-49e0ee4c7942","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:b339b16df2f4186d43e2e94ffc35c1c34a453b5eb66887f644ea27ade2839d01","observation_id":"3281a1a2-3759-4e31-a6de-dc46c4f75653","resolution":{"observed_at":"2026-06-28T17:22:24.823056Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12118","last_updated":"2025-02-18T18:54:12Z","snapshot_observed_at":"2026-08-03T17:17:48.809885Z","submitted_at":"2025-02-17T18:43:24Z","title":"Scaling Test-Time Compute Without Verification or RL is Suboptimal","version":2},"cited_work":{"arxiv_id":"2502.12118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.12118","snapshot_observed_at":"2026-07-03T20:38:56.081466Z","title":"Scalingtest-timecomputewithout verification or rl is suboptimal","venue":null,"work_id":"f01d383c-eb52-4fe5-8e17-9f763cf82d39","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2502.12118","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:8a2129349e38fdb454e6db5d9ca29b5c588b9ffde5706db5c76b90a578a25892","observation_id":"959bee1d-0987-4be4-adfc-c3212bae2d98","resolution":{"observed_at":"2026-06-28T17:22:24.924302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.21444","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T04:19:34.001123Z","title":"Can large reasoning models self-train?","venue":null,"work_id":"676b6b60-dcea-400f-9a21-469309587fd2","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:73317dad655e124ab3fa1a90ef33ef0b75fdf93a2f58e79b171d710bfaa88a7d","observation_id":"f03008c7-d87b-4c6a-af33-17a8242f9f23","resolution":{"observed_at":"2026-06-28T17:22:24.846010Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.10947","last_updated":"2026-02-25T01:06:05Z","snapshot_observed_at":"2026-07-30T09:54:40.100382Z","submitted_at":"2025-06-12T17:49:55Z","title":"Spurious Rewards: Rethinking Training Signals in RLVR","version":2},"cited_work":{"arxiv_id":"2506.10947","doi":"10.3390/app14041521","metadata_source":"pith","pith_arxiv_id":"2506.10947","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Spurious Rewards: Rethinking Training Signals in RLVR","venue":"cs.AI","work_id":"8e05ef02-44f0-41ce-aea5-d954f72e9546","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2506.10947","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:96244e14eda112df1dc1a6119c66b53625ecd4166ae84528867238d8290c349e","observation_id":"03f90375-e681-4bc1-915a-4f83010bb428","resolution":{"observed_at":"2026-06-28T17:22:24.873180Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.07685","doi":"10.48550/arxiv.2511.07685","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"ResearchRubrics : A benchmark of prompts and rubrics for evaluating deep research agents","venue":null,"work_id":"4b4b3d79-0d0f-4c57-9d4e-37ea9fad28af","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:a0ca451d9a54f8068d1fb335fbf90aba9dd5a9a05fd223120b7a51d93808568a","observation_id":"a301ee11-2939-4f69-8dc7-da88abacbfca","resolution":{"observed_at":"2026-06-28T17:22:24.880044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02508","last_updated":"2025-06-16T03:29:47Z","snapshot_observed_at":"2026-07-06T20:31:03.484385Z","submitted_at":"2025-02-04T17:26:58Z","title":"Satori: Reinforcement Learning with Chain-of-Action-Thought Enhances LLM Reasoning via Autoregressive Search","version":3},"cited_work":{"arxiv_id":"2502.02508","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02508","snapshot_observed_at":"2026-07-03T05:57:41.429120Z","title":"Satori: Reinforcement learning with chain-of-action-thought enhances LLM reasoning via autoregressive search","venue":null,"work_id":"66c78192-e23f-4d45-b4d4-e395c45ad50d","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2502.02508","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:e9f22d7579f677e0f0233e603fa867c1b27fe56366286d21d428c8c2fad62706","observation_id":"d65c1290-9258-4ad1-accd-ea2af3017a66","resolution":{"observed_at":"2026-06-28T17:22:24.890131Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02674","last_updated":"2025-02-25T16:59:11Z","snapshot_observed_at":"2026-07-06T20:01:05.921085Z","submitted_at":"2024-12-03T18:47:26Z","title":"Mind the Gap: Examining the Self-Improvement Capabilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2412.02674","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02674","snapshot_observed_at":"2026-07-01T20:56:13.476979Z","title":"Mind the gap: Examining the self-improvement capabilities of large language models","venue":null,"work_id":"13db825e-1742-4503-aff5-6755fdc61624","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2412.02674","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:9a03851e548f59e387e3da5e8e2f5e1fa2fdd3c4294bad36af0bd778f5c90a0a","observation_id":"f5591126-1c6c-4e21-b113-78511853ae51","resolution":{"observed_at":"2026-06-28T17:22:24.928244Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.00075","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T23:36:23.367407Z","title":"arXiv preprint arXiv:2507.00075 , year=","venue":null,"work_id":"2a83a031-5c4c-47db-83b6-9023b522c562","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:4deba2ed25e57e714709fc015cf92162e9e2fb2aeab4703086bfbc584413a286","observation_id":"bf023d46-7256-4c25-ba92-143c70e8fad8","resolution":{"observed_at":"2026-06-28T17:22:24.892779Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"cited_work":{"arxiv_id":"2504.20571","doi":"10.18653/v1/2024.acl-long.643","metadata_source":"pith","pith_arxiv_id":"2504.20571","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","venue":"cs.LG","work_id":"75a5258b-4143-4f2f-99f3-6d950a496305","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2504.20571","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:c638a34a87ab252dcaec4ba70b3e0c76abb5b09537bc722a56877f0715723428","observation_id":"56b0ad6b-cd6a-47cc-877f-7df9ddc22f11","resolution":{"observed_at":"2026-06-28T17:22:24.111531Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12735","last_updated":"2025-04-26T15:42:47Z","snapshot_observed_at":"2026-07-06T19:34:41.712266Z","submitted_at":"2024-10-16T16:51:01Z","title":"CREAM: Consistency Regularized Self-Rewarding Language Models","version":5},"cited_work":{"arxiv_id":"2410.12735","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.12735","snapshot_observed_at":"2026-06-28T17:22:24.881410Z","title":"Cream: Consistency regularized self-rewarding language models.arXiv preprint arXiv:2410.12735, 2024b","venue":null,"work_id":"51b24f5a-760f-4979-830e-ea837ee38458","year":2024},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2410.12735","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:788d3bfd200208448a9e21aedf8e600971105b9d2b6b131a20c47ffac910bdd8","observation_id":"5f9d1c4b-b980-4951-bed9-2d7cad5ab78c","resolution":{"observed_at":"2026-06-28T17:22:24.883038Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.14843","doi":"10.48550/arxiv.2507.14843","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"The invisible leash: Why rlvr may or may not escape its origin","venue":null,"work_id":"71f17397-1cd4-4f68-ab1d-e91bb5f5953d","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:69963db37c89d5b3e4516fa1bfd999de7db971ad7df474675749afd131404944","observation_id":"2d63038d-9174-40a4-9294-783aa9b9a061","resolution":{"observed_at":"2026-06-28T17:22:24.891235Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.23123","last_updated":"2025-03-04T06:22:40Z","snapshot_observed_at":"2026-07-06T19:42:21.306600Z","submitted_at":"2024-10-30T15:31:54Z","title":"On Memorization of Large Language Models in Logical Reasoning","version":2},"cited_work":{"arxiv_id":"2410.23123","doi":"10.48550/arxiv.2410.23123","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.23123","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"On memorization of large language models in logical reasoning","venue":null,"work_id":"79c008b1-cfca-4ce2-ba0b-ae4963d341f9","year":2024},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2410.23123","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:d8cb7208aa98877cd0d644ffb7d6912331b6a3c5d10204f645126126279601e7","observation_id":"205eac9a-7f3f-473f-9b7e-ba3b271e323d","resolution":{"observed_at":"2026-06-28T17:22:24.888547Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:a0cce30a25a18089cbacfcef58ffacc1a49727a802dc28040acc91a3e8ee77b6","observation_id":"17319cef-dde8-4c2c-9d86-28417351a0c9","resolution":{"observed_at":"2026-06-28T17:22:24.875497Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.23751","last_updated":"2025-09-03T14:36:00Z","snapshot_observed_at":"2026-07-31T09:00:01.373564Z","submitted_at":"2025-07-31T17:38:50Z","title":"CoT-Self-Instruct: Building high-quality synthetic prompts for reasoning and non-reasoning tasks","version":2},"cited_work":{"arxiv_id":"2507.23751","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.23751","snapshot_observed_at":"2026-07-04T20:40:08.284654Z","title":"Cot- self-instruct: Building high-quality synthetic prompts for reasoning and non-reasoning tasks.arXiv preprint arXiv:2507.23751, 2025a","venue":null,"work_id":"181a7fd6-f172-4c7e-8068-4cea4b9af3b0","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2507.23751","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:83db2f8c8528ddffe942f62f0d8c0cb13ad284ec435b3fb7af225e5d23fe4199","observation_id":"e315e6d2-d6b1-4853-858b-f75e8fbbf44e","resolution":{"observed_at":"2026-06-28T17:22:24.918750Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.12338","last_updated":"2025-08-17T11:57:34Z","snapshot_observed_at":"2026-07-06T22:14:01.538942Z","submitted_at":"2025-08-17T11:57:34Z","title":"Wisdom of the Crowd: Reinforcement Learning from Coevolutionary Collective Feedback","version":1},"cited_work":{"arxiv_id":"2508.12338","doi":null,"metadata_source":"pith","pith_arxiv_id":"2508.12338","snapshot_observed_at":"2026-07-09T22:56:37.751821Z","title":"arXiv preprint arXiv:2508.12338 , year=","venue":"cs.AI","work_id":"3e6deeb7-caf8-4c3a-9564-38f622876d46","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2508.12338","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:33494a9873cbff7db50d91888d8cf5affff45d97994d2b628d5a70549c8e3a7a","observation_id":"f0e7d95e-baf8-4494-b6b0-69750ca5a0be","resolution":{"observed_at":"2026-06-28T17:22:24.921535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.16175","last_updated":"2026-02-05T18:03:03Z","snapshot_observed_at":"2026-08-01T10:41:29.420510Z","submitted_at":"2026-01-22T18:24:00Z","title":"Learning to Discover at Test Time","version":2},"cited_work":{"arxiv_id":"2601.16175","doi":"10.18653/v1/2025.findings-emnlp.691","metadata_source":"pith","pith_arxiv_id":"2601.16175","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Learning to Discover at Test Time","venue":"cs.LG","work_id":"e6d32375-24bd-4c3a-a7b8-fe20121f31b1","year":2026},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2601.16175","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:2c3aa6852d9565e295dbdb12e2979ff44568a5e8c95fdecab38eea6ab953bcec","observation_id":"20b0da13-b25d-40f3-be3d-edf1d8366c7d","resolution":{"observed_at":"2026-06-28T17:22:24.881981Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.05605","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T17:22:24.920321Z","title":"arXiv preprint arXiv:2502.05605 , year =","venue":null,"work_id":"0427d522-df63-4dd6-a161-d4fc857e8fd1","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:1de1d7b956b5aa82dcb7c1cee953d232dd063190bb31f21c1cf264b80ddf070d","observation_id":"9ca7a1b8-1fae-4c19-a619-8726fa10704e","resolution":{"observed_at":"2026-06-28T17:22:24.921961Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.05280","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T03:45:55.772054Z","title":"arXiv preprint arXiv:2601.05280 , doi=","venue":null,"work_id":"8329fa1f-c9ae-4e5e-8c23-71df7cfcb4cf","year":2026},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:c50b57f6faf26f8b280d2fe6716990856fdcb980ddc88c71ca02170ea3a761f2","observation_id":"c26257f0-6d10-4a9e-ba02-094f8a5eb183","resolution":{"observed_at":"2026-06-28T17:22:24.907486Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05812","last_updated":"2025-05-18T13:41:33Z","snapshot_observed_at":"2026-07-06T21:05:59.227458Z","submitted_at":"2025-04-08T08:48:51Z","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","version":3},"cited_work":{"arxiv_id":"2504.05812","doi":"10.48550/arxiv.2504.05812","metadata_source":"pith","pith_arxiv_id":"2504.05812","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Right question is already half the answer: Fully unsupervised llm reasoning incentivization","venue":"cs.LG","work_id":"307cd976-c17b-46e9-b54f-41fd64b30e5e","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2504.05812","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:7aa465a0b1a7a5d26600541130fe53adb8334c9ad91de3c70c7248ff63fd517a","observation_id":"24bdac98-83ff-46a8-911f-ab6accbe959b","resolution":{"observed_at":"2026-06-28T17:22:24.919029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:12.480934+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:12.480934+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.16084","last_updated":"2025-06-30T15:59:26Z","snapshot_observed_at":"2026-07-06T21:13:13.686703Z","submitted_at":"2025-04-22T17:59:56Z","title":"TTRL: Test-Time Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2504.16084","doi":"10.48550/arxiv.2504.16084","metadata_source":"pith","pith_arxiv_id":"2504.16084","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"TTRL: Test-Time Reinforcement Learning","venue":"cs.CL","work_id":"54ef1983-e8f7-4b17-9b74-6155b56c8b00","year":2025},"citing_paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-28T17:18:37.671369Z"},"links":{"cited_paper":"/paper/2504.16084","citing_paper":"/paper/2606.01075"},"observation_digest":"sha256:4cdb7abc6f657e6298beafd8727eca909e9f2fa99e4fbbf8db194a2f3f0cb106","observation_id":"44d1bcc3-e190-4656-bd72-f09fac76d353","resolution":{"observed_at":"2026-06-28T17:22:24.927167Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2606.01075","last_updated":"2026-06-02T05:50:15Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T07:43:19Z","title":"On the Generalization Gap in Self-Evolving Language Model Reasoning"},"reference_resolution":{"displayed":46,"state_counts":{"malformed_identifier":0,"metadata_mismatch":9,"parse_uncertain":0,"unresolved":4,"verified_exact":33,"verified_fuzzy":0},"total_outbound_references":46},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 46 of 46 outbound references and 0 inbound Pith citation observations for arXiv:2606.01075."}