{"as_of":"2026-08-04T06:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:af58c3c0ec7ff00cb95f1a1d90684dd8a475f1048632ca9a319f5070484e6308","coverage":[{"denominator":80,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":80,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T12:11:00.402276Z","state":"measured"},{"denominator":81,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":81,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-03T06:30:56.289259+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T07:19:00.012926Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-06-30T07:24:21.458766Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"cited_work":{"arxiv_id":"2605.28388","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.28388","snapshot_observed_at":"2026-06-30T07:24:21.458766Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","venue":"cs.AI","work_id":"e3ed4ebf-7d32-44d0-819c-4ce0f48617d1","year":2026},"citing_paper":{"arxiv_id":"2606.30345","last_updated":"2026-06-29T14:20:47Z","snapshot_observed_at":"2026-08-01T20:13:14.564783Z","submitted_at":"2026-06-29T14:20:47Z","title":"DRIFT: Difficulty Routing Self-DIstillation with Rhythm-Gated Exploration and Success BuFfer Training","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T07:19:00.012926Z"},"links":{"cited_paper":"/paper/2605.28388","citing_paper":"/paper/2606.30345"},"observation_digest":"sha256:1d790e952b6806fa749e86d70d94c95eb58264ce8b4dfc9b2f4de0f5daca86a1","observation_id":"12ba30fa-be86-4e36-9321-482ee6e65a26","resolution":{"observed_at":"2026-06-30T07:24:21.461392Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.28388/citation-record","integrity":"/paper/2605.28388/integrity","json":"/paper/2605.28388/citation-record.json","paper":"/paper/2605.28388"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.14740","last_updated":"2024-02-26T18:26:25Z","snapshot_observed_at":"2026-07-06T17:34:07.737296Z","submitted_at":"2024-02-22T17:52:34Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","version":2},"cited_work":{"arxiv_id":"2402.14740","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.14740","snapshot_observed_at":"2026-07-09T08:56:06.435604Z","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","venue":"cs.LG","work_id":"7bb8f9ec-1241-4472-a4fa-c636c6d79892","year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2402.14740","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:0b9cbe6c7aa15f7bb2965ef5e714a6ad9d7a6722f9b8fc8fddc03ea7c582f80e","observation_id":"fab16154-6f0c-4db4-a2c9-003d60c9dece","resolution":{"observed_at":"2026-06-29T12:13:26.556805Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Online difficulty filtering for reasoning oriented reinforcement learning","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:34d05f35d037c62e55234c53f1daff2e37baa791bf749c6725d417792e4e89d8","observation_id":"8f908df6-1388-49e4-9ebc-5ee16e0550c3","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Curriculum learning","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:a8e03def3e5324ddeba0e3a48fcce23cc10c39bee94ecbb8c393636ac748d5b0","observation_id":"811a0bf9-f40a-47e3-83bc-1897ca42416d","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Temporal sparse autoencoders: Leveraging the sequential nature of language for interpretability","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:b5874cc536ffdb4f63a073f4db1a119d6c88101d24d0b07c88d65668a4f9fd3d","observation_id":"2503259f-b4a7-43d2-aa49-00eb34bf1e91","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.18207","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T21:07:23.923615Z","title":"Hansen and Duo Peng and Yuhui Zhang and Alejandro Lozano and Min Woo Sun and Emma Lundberg and Serena Yeung-Levy , year=","venue":null,"work_id":"24cdde52-11e0-4409-8656-4a46c92a8552","year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:0b1b2889620848af4c2f6985d695a42e0ecfdcc0b5e641e20cd15dba3dfd7b2f","observation_id":"97074281-0cc8-47ca-9187-5781ab93cb35","resolution":{"observed_at":"2026-06-29T12:13:26.572974Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"cited_work":{"arxiv_id":"2502.01456","doi":"10.48550/arxiv.2502.01456","metadata_source":"pith","pith_arxiv_id":"2502.01456","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Process Reinforcement through Implicit Rewards","venue":"cs.LG","work_id":"c31a2126-86f9-44f3-91f3-208d0fc1463a","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2502.01456","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:b5cdf23504a1dee96d6fc4c341d6d2e3e8b03bb14da9d307d03e0dc5594a4257","observation_id":"b3d2478d-512b-4561-9eea-9c11b551aafe","resolution":{"observed_at":"2026-06-29T12:13:26.606024Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15260","last_updated":"2025-08-21T05:48:38Z","snapshot_observed_at":"2026-08-02T03:10:09.020351Z","submitted_at":"2025-08-21T05:48:38Z","title":"Deep Think with Confidence","version":1},"cited_work":{"arxiv_id":"2508.15260","doi":"10.48550/arxiv.2508.15260","metadata_source":"pith","pith_arxiv_id":"2508.15260","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Deep Think with Confidence","venue":"cs.LG","work_id":"39c40c74-9f55-406d-b67a-37d9b88aa6b3","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2508.15260","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:e0f5685c221803a1447b15c07624624d583a5ba97069f49be23ae6fd1aec1071","observation_id":"4af94b0d-0415-4649-8264-9320efb95736","resolution":{"observed_at":"2026-06-29T12:13:26.579988Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"I have covered all the bases here: Interpreting reasoning features in large language models via sparse autoencoders","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:6e3b1e7858482bb34eceb5c90506ff9b8f428aaf5a0ac95e55cd4f8403deebb4","observation_id":"e379eb8a-8ec5-433e-93e6-f4dd04041912","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Scaling and evaluating sparse autoencoders","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:9be721bf6aef2e28fe2ebcb270de6ffea07c53b20d6bd010f53341287b92aed3","observation_id":"10a882b1-b1ed-4733-8ee0-e1a7b1d76650","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Olympiadbench: A challenging benchmark for promoting agi with olympiad-level bilingual multimodal scientific problems","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:fda326a95e4376aa88d496ea6827d75da88ba76646f4e24ba87a55e6c1739ce0","observation_id":"aa457bc8-42cb-4367-aaa6-baba872204b7","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-07-10T16:57:24.565388Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:cfeb90b351d004ef4846df1bc0a49781e410e67181369bce35d95942f2de514e","observation_id":"6f13bddb-134b-4e26-9029-79f28e8ba273","resolution":{"observed_at":"2026-06-29T12:13:26.568125Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03262","last_updated":"2025-11-10T15:11:13Z","snapshot_observed_at":"2026-08-02T05:27:47.490711Z","submitted_at":"2025-01-04T02:08:06Z","title":"REINFORCE++: Stabilizing Critic-Free Policy Optimization with Global Advantage Normalization","version":9},"cited_work":{"arxiv_id":"2501.03262","doi":"10.48550/arxiv.2501.03262","metadata_source":"pith","pith_arxiv_id":"2501.03262","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"REINFORCE++: Stabilizing Critic-Free Policy Optimization with Global Advantage Normalization","venue":"cs.CL","work_id":"557f9e99-cb00-4dd2-92fd-67ddcddbb35d","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2501.03262","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:ebdee4d0a33388714bc9f5fede021d4f539be9b62e1d5b1c49674c200269017c","observation_id":"d9ea390b-d186-4cc7-8ebd-be33fdad455c","resolution":{"observed_at":"2026-06-29T12:13:26.528934Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:52:56.973979+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:52:56.973979+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Sparse autoencoders find highly interpretable features in language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:629394786454cfe708ae9141bc34b7c15af785de2c2f5b1fb87bca0e792b3123","observation_id":"ea7dbd3e-77ad-400a-9ef3-94050c5b047c","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.19803","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T18:30:01.575162Z","title":"Vcrl: Variance-based curriculum reinforcement learning for large language models","venue":null,"work_id":"87da4de6-b94c-43ee-8a28-bd43363bea62","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:721da3a5bc83a72bc7e95907ba77bd8eba4b9e676c554b89a728406b78aeb47a","observation_id":"0bbe7b15-9844-46b8-8d32-3be1876227e0","resolution":{"observed_at":"2026-06-29T12:13:26.570523Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15124","last_updated":"2025-04-14T22:39:09Z","snapshot_observed_at":"2026-07-06T19:55:37.400185Z","submitted_at":"2024-11-22T18:44:04Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","version":5},"cited_work":{"arxiv_id":"2411.15124","doi":"10.48550/arxiv.2411.15124","metadata_source":"pith","pith_arxiv_id":"2411.15124","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Tulu 3: Pushing Frontiers in Open Language Model Post-Training","venue":"cs.CL","work_id":"28c9dbea-056a-48c2-8000-85f809827e45","year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2411.15124","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:da9c48f2c5b1121a68a9dca27785058170d9fa25688f8e53303fcbc3e3c6553a","observation_id":"a61e86aa-afd9-4c73-a3b9-d23b90cc6996","resolution":{"observed_at":"2026-06-29T12:13:26.597014Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:00.522112+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Le, Myeongho Jeon, Kim Vu, Viet Dac Lai, and Eunho Yang","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:ac456fabe5c03d34bcd9753b6a9bf2722771ea9ec17a2de2b70b07c6f7c334e7","observation_id":"b20825b9-6f6c-4197-a99b-a4ef5a8c4463","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Solving quan- titative reasoning problems with language models.Advances in neural information processing systems, 35:3843–3857, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:e85c37deac73426686a76d3ad2534bf41c344d29c666fca56fd31e3f1c642672","observation_id":"1217f1ce-52bb-4cdb-a40f-37884b290287","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Numinamath: The largest public dataset in ai4maths with 860k pairs of competition math problems and solutions.Hugging Face repository, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:179d121137ee3699b3a6caddff02601e625c4a59e0ae1284b7b512e7bcfe4d92","observation_id":"b828254b-15b1-4e9b-9e49-a67b1fe35a71","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"QuestA: Expanding reasoning capacity in LLMs via question augmentation","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:07c3a935d9cc98f4413babee099f8caf368223d4d2d2da83dfec64ead2c3267b","observation_id":"92fb137b-6598-4706-8eba-adf896fa57cc","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10505","last_updated":"2024-05-16T02:22:23Z","snapshot_observed_at":"2026-07-06T16:33:55.369704Z","submitted_at":"2023-10-16T15:25:14Z","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","version":4},"cited_work":{"arxiv_id":"2310.10505","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10505","snapshot_observed_at":"2026-07-04T08:09:40.661453Z","title":"arXiv preprint arXiv:2310.10505 , year=","venue":null,"work_id":"de1cbb7b-6798-4667-8e51-9c68ae61551b","year":2023},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2310.10505","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:9fac761f0e3e91082b385a404ef0e4b67f8a7cffdd6f8f63cfd558925b59aad7","observation_id":"87373631-0bff-4bec-8c62-2e845eab17ec","resolution":{"observed_at":"2026-06-29T12:13:26.588012Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.14029","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T18:17:33.741584Z","title":"Beyond pass@ 1: Self-play with variational problem synthesis sustains rlvr.arXiv preprint arXiv:2508.14029","venue":null,"work_id":"422e2c72-2ab8-40b1-b568-8a4158268572","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:797dc801e1a3af49c58aef3b97c334441a1116fc2c83d7ca9819334dbcad3c3e","observation_id":"d76253e8-46e5-45e5-a2ef-fbbaa87f45a7","resolution":{"observed_at":"2026-06-29T12:13:26.546069Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05147","last_updated":"2024-08-19T07:51:05Z","snapshot_observed_at":"2026-08-01T05:08:19.149239Z","submitted_at":"2024-08-09T16:06:42Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","version":2},"cited_work":{"arxiv_id":"2408.05147","doi":"10.48550/arxiv.2408.05147","metadata_source":"pith","pith_arxiv_id":"2408.05147","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","venue":"cs.LG","work_id":"90aa1ebe-ab2b-4462-adfd-047832493655","year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2408.05147","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:4ace89462d19afd41aa427dba17a0736d78844ab5fb6475844c00e02d1c01a71","observation_id":"7ec33387-212e-4c05-a071-246ab361e2e8","resolution":{"observed_at":"2026-06-29T12:13:26.551175Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-07-31T17:40:45.057417Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":"2305.20050","doi":"10.1007/bf00262952","metadata_source":"pith","pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Let's Verify Step by Step","venue":"cs.LG","work_id":"6d05b790-04c5-4fd2-91b2-ba1dfdd5770f","year":2023},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:7992580084d8b68819d014bd92c7a8fc56855f85ad0e0448da931f3946e823c8","observation_id":"5aede3d3-09e1-43b6-b5e4-a483ae4db0e2","resolution":{"observed_at":"2026-06-29T12:13:26.603752Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05470","last_updated":"2025-10-27T09:57:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-08T17:58:45Z","title":"Flow-GRPO: Training Flow Matching Models via Online RL","version":5},"cited_work":{"arxiv_id":"2505.05470","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.05470","snapshot_observed_at":"2026-07-09T02:25:55.898592Z","title":"Flow-GRPO: Training Flow Matching Models via Online RL","venue":"cs.CV","work_id":"bf1e8e81-ff31-401a-a5dc-d9c49df168ab","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2505.05470","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:333b31b9a1a4f5ac4a6b7d75ad97cd68f671061a5c832a0c95db0f1c81d09f9f","observation_id":"5896e8dc-729b-4681-ab04-c871df99d358","resolution":{"observed_at":"2026-06-29T12:13:26.533838Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":"2503.20783","doi":"10.48550/arxiv.2503.20783","metadata_source":"pith","pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","venue":"cs.LG","work_id":"ec354f3b-9484-4a0c-94c8-92d4d0260835","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:fdbce55c9f8b69f83c1970b95e2cb808f39723346fac7066cd3ff8e1b437d54d","observation_id":"0eeb5c27-d76b-405c-a8b8-b3496223dac5","resolution":{"observed_at":"2026-06-29T12:13:26.590325Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.21877","last_updated":"2026-05-07T15:50:49Z","snapshot_observed_at":"2026-07-07T11:16:59.422932Z","submitted_at":"2026-03-23T12:08:47Z","title":"P^2O: Joint Policy and Prompt Optimization","version":3},"cited_work":{"arxiv_id":"2603.21877","doi":null,"metadata_source":"pith","pith_arxiv_id":"2603.21877","snapshot_observed_at":"2026-06-29T12:13:26.562553Z","title":"P^2O: Joint Policy and Prompt Optimization","venue":"cs.LG","work_id":"0cd5e33b-59de-4925-8ba5-9bdc3f8d9241","year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2603.21877","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:341232afca464806bb851591a1b329ef8b912ebda036e7a75124fcafe4c338ad","observation_id":"9fd02390-6969-4edf-87dd-88317e7426e9","resolution":{"observed_at":"2026-06-29T12:13:26.563632Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Deepscaler: Surpassing o1-preview with a 1.5 b model by scaling rl.Notion Blog, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:01006997c1c791a4a37286d34325da29cea2c4e950d764c45fb702971302b19a","observation_id":"8be2425e-b6f8-4692-9669-309b22e8d798","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Bissyande, Haoye Tian, and Bach Le","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:d2be659b4012728795947a05b34fc776cd07d08b5bcd3c69eab24c610c7bf82d","observation_id":"a6a43801-b238-4adb-b480-1e8da677afaa","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Michaud, Yonatan Belinkov, David Bau, and Aaron Mueller","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:929305f5044930469da38771c98cf5aa950a36a7b80298adc5e30a63f718c6ba","observation_id":"fbc7b728-24cb-416e-bd52-b6822afa9937","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Sparse autoencoder.CS294A Lecture notes, 72(2011):1–19, 2011","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:aaec66be4eae808b563d12085d2e502a6e16c401ca86f64dde0d48b4d4cc3bef","observation_id":"667c6f5a-0239-46bb-963a-cb36ddb04a28","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19634","last_updated":"2025-03-19T13:55:33Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:57:34Z","title":"MedVLM-R1: Incentivizing Medical Reasoning Capability of Vision-Language Models (VLMs) via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2502.19634","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19634","snapshot_observed_at":"2026-07-04T03:39:30.318199Z","title":"Advances in neural in- formation processing systems, 35:27730–27744","venue":null,"work_id":"ab838cb8-f672-4c40-84f6-a7b28d69e657","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2502.19634","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:e45e071ff8782d61b1604101b3f15ec8ab2cb1f952cfcf7f709a6c950f71a471","observation_id":"52ba3a10-6510-4da7-8117-e8f853cc6d58","resolution":{"observed_at":"2026-06-29T12:13:26.541175Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.06632","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T08:09:40.613456Z","title":"Curricu- lum reinforcement learning from easy to hard tasks im- proves llm reasoning.arXiv preprint arXiv:2506.06632","venue":null,"work_id":"92cf1fbf-7715-4a04-a730-0b38da9c54f1","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:38804e4485181d6e51e56da30ac84cec7e2a3ac86888bdc8b91eb71a92ed4129","observation_id":"18a35d8f-1f77-455b-a899-cfadd56af530","resolution":{"observed_at":"2026-06-29T12:13:26.553644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Automatically interpreting millions of features in large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:ef5483f280a8d0b7d8115c04c8df3a4eb51d50f9e560901836e3e7ea73e6af42","observation_id":"a3a1bd52-5811-46ab-9fda-09792f946bde","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.20733","last_updated":"2026-04-22T16:20:41Z","snapshot_observed_at":"2026-07-06T23:07:25.185196Z","submitted_at":"2026-04-22T16:20:41Z","title":"Near-Future Policy Optimization","version":1},"cited_work":{"arxiv_id":"2604.20733","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.20733","snapshot_observed_at":"2026-07-02T01:36:25.013114Z","title":"Near-Future Policy Optimization","venue":"cs.LG","work_id":"e243c163-4b02-4331-af79-f9d5dc68c65a","year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2604.20733","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:6ee0ad830f23a54c3049733a4260d4d69a09da5f485c0cd3a2b50391512a6f9a","observation_id":"09290c32-bce7-428c-ae22-1d6cf10e6b9c","resolution":{"observed_at":"2026-06-29T12:13:26.592555Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:9e8e62f564de30cff9913b5ded5980ba289c26f295c6d773cdcbef6a32699093","observation_id":"6c78e222-cf37-476f-acbe-9109267146b0","resolution":{"observed_at":"2026-06-29T12:13:26.599172Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:c2f58b3ceaccc2017cff95d13121fda52c5523bb53f0e6cc075b03bbd93c484e","observation_id":"7c21adc3-97e9-4ab3-8285-7e2ffd094ea2","resolution":{"observed_at":"2026-06-29T12:13:26.582443Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"cited_work":{"arxiv_id":"2409.19256","doi":"10.1145/3689031.3696075.url:","metadata_source":"pith","pith_arxiv_id":"2409.19256","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","venue":"cs.LG","work_id":"7eb9c9f4-b322-4bba-8011-09ff8d6ad801","year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2409.19256","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:b05cc3199ab6a41c66df7c2a0831579ca9067bc1b5d750318e992f0bcf362674","observation_id":"12fa6662-c969-417f-9940-e15496ced6a2","resolution":{"observed_at":"2026-06-29T12:13:26.561427Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Towards high data efficiency in reinforcement learning with verifiable reward","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:18fbf077fb8fa7db8daacffb09e59486fc134c1e17fa189973f268d0e831cb6f","observation_id":"d3f621c5-34cc-4a4a-bbc8-a404eb19a703","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20571","last_updated":"2025-10-24T10:02:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-29T09:24:30Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","version":3},"cited_work":{"arxiv_id":"2504.20571","doi":"10.18653/v1/2024.acl-long.643","metadata_source":"pith","pith_arxiv_id":"2504.20571","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","venue":"cs.LG","work_id":"75a5258b-4143-4f2f-99f3-6d950a496305","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2504.20571","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:435d64766ab36729cef89bd76df3a6b12af2261a009c9fb45e278f2fc61de1ce","observation_id":"e13180f9-a5d3-4ef0-889f-94b394a5a387","resolution":{"observed_at":"2026-06-29T12:13:26.565970Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.01223","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T16:24:01.012815Z","title":"Learn hard problems during rl with reference guided fine-tuning","venue":null,"work_id":"cca99282-a0f6-4e3b-894f-0589afedb0de","year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:c1081f0aff19f8d92a4636de040dbe1438323d59e621e9a7fcce72b0c7be676e","observation_id":"772734fd-d24c-4b8f-9fe8-3444bb2043c6","resolution":{"observed_at":"2026-06-29T12:13:26.548687Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07818","last_updated":"2025-08-28T17:19:45Z","snapshot_observed_at":"2026-07-31T14:51:03.625964Z","submitted_at":"2025-05-12T17:59:34Z","title":"DanceGRPO: Unleashing GRPO on Visual Generation","version":4},"cited_work":{"arxiv_id":"2505.07818","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.07818","snapshot_observed_at":"2026-07-09T03:05:55.316091Z","title":"DanceGRPO: Unleashing GRPO on Visual Generation","venue":"cs.CV","work_id":"7404dd36-8f9c-478f-b089-ef9f8189c711","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2505.07818","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:2ff7fe631a9dc620d59bce7a99545d4c7ae54dd8e9be9b8e7d25c06dd374c32e","observation_id":"cce9635e-cd79-4155-bfc7-8c52124843ad","resolution":{"observed_at":"2026-06-29T12:13:26.584739Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-07-06T19:17:41.512834Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":"2409.12122","doi":"10.18653/v1/2025.emnlp-main.712","metadata_source":"pith","pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","venue":"cs.CL","work_id":"a097c5d4-6d32-46ee-9826-57d532bbfc9c","year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:4abb1b391d2e19dd393c9c7e40a55517e1340b3dc4a1850e62a1a2e8747c70c6","observation_id":"6ab60fd2-f140-4963-a446-84a35bdc8f18","resolution":{"observed_at":"2026-06-29T12:13:26.601447Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02333","last_updated":"2025-09-08T14:50:44Z","snapshot_observed_at":"2026-08-04T00:45:58.176259Z","submitted_at":"2025-09-02T14:01:07Z","title":"DCPO: Dynamic Clipping Policy Optimization","version":2},"cited_work":{"arxiv_id":"2509.02333","doi":"10.48550/arxiv.2509.02333","metadata_source":"pith","pith_arxiv_id":"2509.02333","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Dcpo: Dynamic clipping policy optimization","venue":"cs.CL","work_id":"8c19db85-51aa-4d83-8ade-6f3bc8d1c75e","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2509.02333","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:4162e59e33cf0c8b7f14f33f3615688abf6584b86654d79e37c0547cb8d724c4","observation_id":"ac7a7c67-77f7-46e8-9875-4e3aaeeb8c81","resolution":{"observed_at":"2026-06-29T12:13:26.575383Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.13755","last_updated":"2026-04-12T13:17:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-19T11:51:40Z","title":"Depth-Breadth Synergy in RLVR: Unlocking LLM Reasoning Gains with Adaptive Exploration","version":8},"cited_work":{"arxiv_id":"2508.13755","doi":null,"metadata_source":"pith","pith_arxiv_id":"2508.13755","snapshot_observed_at":"2026-07-04T08:09:40.621610Z","title":"Depth-Breadth Synergy in RLVR: Unlocking LLM Reasoning Gains with Adaptive Exploration","venue":"cs.LG","work_id":"f6699fce-0a85-4d3e-958a-b94201b37151","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2508.13755","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:329138ee963df9037237e69897a047e32916b13a9859d66d5ff585df966c8811","observation_id":"4c81fe84-6e9c-40d8-bf94-418c9af1ee44","resolution":{"observed_at":"2026-06-29T12:13:26.577872Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Kwok, Zhenguo Li, Adrian Weller, and Weiyang Liu","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:d8f3350270fcde85dac1f40eb2b46de947140c2b88d6b9173c41fb59d07d4a8d","observation_id":"7bead28b-7bf3-4309-87eb-e755ab489a84","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-07-11T03:07:50.815080Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:b74f6fd57ca6e99f0c9f9ba257a2642c7d085a392bacccaa0d335667bf6b75e4","observation_id":"b115c98e-8723-4179-a9b5-1424a4da029b","resolution":{"observed_at":"2026-06-29T12:13:26.594722Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":"2504.13837","doi":"10.48550/arxiv.2504.13837","metadata_source":"pith","pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-07-10T22:47:36.894457Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","venue":"cs.AI","work_id":"d854765a-e664-41c0-8655-21c4bf2e0cc4","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:3134a9ae88b6c6ec5ba8dd159939667d84a3c8d6ca83f5aff3527cfb745d4e16","observation_id":"f616f55a-bff7-4913-acb3-68dcb2fbf2a3","resolution":{"observed_at":"2026-06-29T12:13:26.531518Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05118","last_updated":"2025-04-11T02:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T14:21:11Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2504.05118","doi":"10.1109/access.2024.3384487","metadata_source":"pith","pith_arxiv_id":"2504.05118","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","venue":"cs.AI","work_id":"c2351652-65f7-47cd-ae80-dbcd72a6eb20","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2504.05118","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:cb30a22b0bf8ab7f66c6f2921e14378b6f93c36f0d00c3d0b0f0331272b189e9","observation_id":"5bf5ee62-0083-4b7f-b626-20ca878ee647","resolution":{"observed_at":"2026-06-29T12:13:26.559111Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Wong, and Yu Cheng","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:e4dcecdb3fe0259c1dec7a6aada75ba02cb8c825df5a023a6a0d6d70f7dc7731","observation_id":"a420a429-b884-4e5f-a940-58999437a1ea","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Scaf-GRPO: Scaffolded group relative policy optimization for enhancing LLM reasoning","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:a95b612516829fef89b2b54224be6f97f5457dea5d38672f323131a19f8eda10","observation_id":"76e2995d-9f59-4a95-b381-3755081e7a75","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.03335","last_updated":"2025-10-16T08:23:36Z","snapshot_observed_at":"2026-07-06T21:19:34.329442Z","submitted_at":"2025-05-06T09:08:00Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","version":3},"cited_work":{"arxiv_id":"2505.03335","doi":"10.48550/arxiv.2505.03335","metadata_source":"pith","pith_arxiv_id":"2505.03335","snapshot_observed_at":"2026-07-10T18:17:33.800267Z","title":"Absolute Zero: Reinforced Self-play Reasoning with Zero Data","venue":"cs.LG","work_id":"b59092c4-76ed-4c78-9006-312bde2e40a6","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2505.03335","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:00f41328c2b83c4f723e3013f5855289a60a16df36e70692009488ca394d0b28","observation_id":"5245ac71-be60-4076-8529-1dbe5e835795","resolution":{"observed_at":"2026-06-29T12:13:26.536180Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:52:59.097203+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:52:59.097203+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18071","last_updated":"2025-07-28T11:11:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-24T03:50:32Z","title":"Group Sequence Policy Optimization","version":2},"cited_work":{"arxiv_id":"2507.18071","doi":"10.48550/arxiv.2507.18071","metadata_source":"pith","pith_arxiv_id":"2507.18071","snapshot_observed_at":"2026-07-10T14:27:07.145784Z","title":"Group Sequence Policy Optimization","venue":"cs.LG","work_id":"3a98b53b-9f52-4d95-adf7-89353c0a9a65","year":2025},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"cited_paper":"/paper/2507.18071","citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:b0bbdbae7cdc18af8aa856c9a43c1b6ee782448d66c75efb0ec5787f7d1d423c","observation_id":"a8e5c49b-05a9-4933-ba14-f6e0deacab65","resolution":{"observed_at":"2026-06-29T12:13:26.543437Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"AbsTopK: Rethinking sparse au- toencoders for bidirectional features","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:7a79e6ddfe130c33a3ca0b5a992421b18c13e542d98937aa70066ec33c0c4113","observation_id":"3a9aeec3-b49f-4057-886a-18c342bd32b4","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:d06e22b65696c0492ba52c70f2b9ccb869e83b7f44ed869540c8279224d5f0e3","observation_id":"7fa1a272-ff80-44a3-a4df-9bd6145b082c","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Therefore, she doesn ’t need to do any more situps on Wednesday to meet her goal","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:abb9e416721ea2ec526d35d2800925a29b074e32a29a288d08aca223f57b788e","observation_id":"5eee3410-40fe-4f4a-aa05-a375bb0bfc36","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"\\boxed{{{situps_needed_wednesday}}}","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:483994b52010c6d6ff846b333ba2aff745db92c7527ea880c3429b1acfb42fe3","observation_id":"9da6bb6c-4acb-4868-aa2c-c881a6748d18","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:9b65f72412f176d8ef42ede5352e8439d1023f6883cc7e4f2ccee8e0b7b6ac4d","observation_id":"fd234732-25bc-4b90-9e65-73c2fe106134","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:5517d3e8b99b53186bbde56131f88ffd6dda517ffa113e5305fbc8868dc18f16","observation_id":"98bcb9f6-b4f5-4091-a57e-a4a23cc8ace1","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:a9d5df8a4b5e10f58997632cc20b59a3a1a8d2ba3ce3ea10d09e780e00b09c51","observation_id":"ec722431-3a22-4b1d-a21b-56722b943835","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:82474a9ee8dac7382f5b5d333c9102fe4de8e23df53be75da67a53acb03b4275","observation_id":"214368da-ef18-40e5-821f-e77fbb55afea","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:5cd183c809eb2cfa2ad175abde1d030fe82d4c594efe499697d729ebd43a8869","observation_id":"001fe2f7-393f-4f73-b181-8689b3cdbde7","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"\\boxed{{{int(total_cost)}}}","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:ee4eedf0f50acef328fceb25a6dc59dcb0f770969cb7a829d66bfa99157db300","observation_id":"3f098db9-22ef-47b5-a92f-85f64dcae12c","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"With 3 foxes, the total number of weasels caught per week is \\(3 \\times 4 = 12\\) weasels, and the total number of rabbits caught per week is \\(3 \\ times 2 = 6\\) rabbits","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:c56b829d4c86144ff31ebe855a885123620c5e3f9d1fca218d524bef461febce","observation_id":"a74382fa-f50b-4c3d-87a8-b07631137756","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:95db4ea7c4ac13ad25172692522ad2606a700cbedff99d34dd1b163e92b1804f","observation_id":"84bee187-b2b4-4b9b-a8aa-6fc286268134","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:57e3e8433019a8d079eb3c5f159e6026fef778c18d80ea1979e23577fadfcfbe","observation_id":"e585aa7f-0d4e-480f-aef3-c12c544c0b87","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:158cce785c9e19364f67de6b5c73c64ee33a8e92f600df804434e0187b3a7ed9","observation_id":"750ba8c1-6af6-4d95-bf92-d65b09734d8e","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"\\boxed{{{int(weasels_left)} {int(rabbits_left)}}}","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:c834201991cf44ed9de8f295180c56474bd6dd35a1f399072afc3c75aefc67ce","observation_id":"9d5f5a23-b6ee-4597-bdbc-c34ca34a57af","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:6de389d62e6a90156d56a99b67cac3103c363e510c855fa73d56b38a7107bbca","observation_id":"567fa6ea-d752-44c0-938c-1d1c08ef4da4","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:bdc041424e53cd45f6298efcbc206b88cae676f43f20f211d3904bdd42ec5973","observation_id":"fb72ff17-6878-4083-aaa5-5545091c8782","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"\\boxed{{{int(difference)}}}","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:23b222619e03d3e6d27b41421ecccfbdf5ba3907d5d55dcb1287f3cfe3b3ba09","observation_id":"49b96d98-745f-4d89-ac84-609849418322","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"So, \\( T = A - 20 \\)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:8b6d3b7f83f30b8597be103e532f12c89f04db3c1a274149450c36e63a5182b1","observation_id":"173b2a80-0414-41ef-9c42-4e80ee70bc2d","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"So, \\( S = M + 10 \\)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:d3450a0de568a85b65d304e08fb1b826f364d785cf184abb26c13f7ef878d3e8","observation_id":"f9e6f6d3-2e7f-4f1a-a1d4-2ce30f7c0d1c","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"So, \\( M = 70 \\)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:c493fe28eb0d07b61fedbe3c4f29f7c095802d5e7de62ed9b3f87c7745a35080","observation_id":"7c0242f4-a184-4ebd-b8c1-f1b89c43b2dd","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"3321.6916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:13:26.537310Z","title":"\\boxed{{{int(total_marks)}}}","venue":null,"work_id":"8feb14e2-2ea5-4532-ba16-fd3fdf95fef0","year":2024},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:763f8f24a470f7a0f4565655a91d221baf37372a71c3c41226971a0605a8ca00","observation_id":"7f7fbaf8-7f0f-49d7-bce2-e4f67e73ac61","resolution":{"observed_at":"2026-06-29T12:13:26.538591Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Suppose that a+ (1/b) and b+ (1/a) are the roots of the equationx 2 −px+q= 0","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:607fef9fe2b9136e3b3f5d5a6a45a23d24f33edd1dbd912335df4c7dd98d9d0c","observation_id":"18939218-ca36-4710-822a-b59de9c1aafa","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Suppose that a+ (z/b) and b+ (1/a) are the roots of the equationx 2 −px+q= 0","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:006655583df4b3a1cdb8cb108ae377f6f9e5fd3efad36604c957c823a2caef11","observation_id":"d761ae3c-a730-4c3e-9b46-25ab600e2630","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Suppose that a+ (1/b) and b+ (z/a) are the roots of the equationx 2 −px+q= 0","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:9d28d79092426d989713455d5209897732114405f78b27b5a5cedbd6f2beb955","observation_id":"54dc86fa-7e99-4043-877c-ecfeb4b11693","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Suppose that a+ (1/b) and b+ (1/a) are the roots of the equation x2 −px+q= 0","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:fab7c8d772f8ab9106c21babfea9e122f873f979379e4ac0b0b500f4c95b604c","observation_id":"ec053849-7c0e-4b77-8036-b50f27b450f2","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Suppose that a+ (z/b) and b+ (1/a) are the roots of the equation x2 −px+q= 0","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:300f458dab8c522417e54f29babeedede5b9e6dc44c6be832f7082ce492912d5","observation_id":"6a9ddd17-f0bf-43d6-aa68-e8129206045e","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T12:11:00.402276Z","title":"Suppose that a+ (1/b) and b+ (z/a) are the roots of the equation x2 −px+q= 0","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-06-29T12:11:00.402276Z"},"links":{"citing_paper":"/paper/2605.28388"},"observation_digest":"sha256:594b0bbcf512949f65dd0b9272319a3f519c213de6850fbc7cc4bd6511257514","observation_id":"03c22ac2-0321-4743-9fc4-0974f10285cf","resolution":{"observed_at":"2026-06-29T12:11:00.402276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2605.28388","last_updated":"2026-05-27T12:25:57Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T23:37:59.285884Z","submitted_at":"2026-05-27T12:25:57Z","title":"Mechanistically Interpreting the Role of Sample Difficulty in RLVR for LLMs"},"reference_resolution":{"displayed":80,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":47,"verified_exact":30,"verified_fuzzy":0},"total_outbound_references":80},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 80 of 80 outbound references and 1 inbound Pith citation observation for arXiv:2605.28388."}