{"as_of":"2026-07-20T17:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:20472a04f55ffcc70208a03bdf539abeca2d36c12b93f386c9b89ca23a2464c1","coverage":[{"denominator":20,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T11:50:26.030339Z","state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-07-20T06:30:07.809122+00:00","state":"measured"},{"denominator":13,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":13,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-09T22:18:13.418579Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T22:26:37.216917Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2602.15620","last_updated":"2026-05-25T12:27:41Z","snapshot_observed_at":"2026-07-06T22:46:12.384926Z","submitted_at":"2026-02-17T14:46:48Z","title":"STAPO: Stabilizing Reinforcement Learning for LLMs by Silencing Rare Spurious Tokens","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-15T21:41:48.690125Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2602.15620"},"observation_digest":"sha256:95a3b1ada78fc89167280a76627c4f9354e50cf0617c73dae490f2e9ac5fa4e5","observation_id":"88d8db98-04ae-413c-a74f-d520c54c0249","resolution":{"observed_at":"2026-05-20T00:00:25.356150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2604.18530","last_updated":"2026-05-27T05:59:47Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:26:00Z","title":"OGER: A Robust Offline-Guided Exploration Reward for Hybrid Reinforcement Learning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T04:29:21.897215Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2604.18530"},"observation_digest":"sha256:973412c69cec514b4c6f271716f039afca93cd458db60ecc320ec1431aa4f703","observation_id":"5d6c6851-6cd5-4edc-8f6b-73d91282500b","resolution":{"observed_at":"2026-05-20T00:00:25.356150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2604.18578","last_updated":"2026-04-30T14:35:55Z","snapshot_observed_at":"2026-07-06T23:05:26.398712Z","submitted_at":"2026-04-20T17:59:01Z","title":"Bounded Ratio Reinforcement Learning","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T04:50:11.020901Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2604.18578"},"observation_digest":"sha256:8e7492bd1b9150bc910809d6f3f25f5053f238a77c124979c223ce79e0999cfd","observation_id":"8c479286-4409-453a-81c8-22345d9703cc","resolution":{"observed_at":"2026-05-20T00:00:25.356150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.04077","last_updated":"2026-04-14T09:48:46Z","snapshot_observed_at":"2026-07-06T23:16:54.673178Z","submitted_at":"2026-04-14T09:48:46Z","title":"Balanced Aggregation: Understanding and Fixing Aggregation Bias in GRPO","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T14:53:35.133157Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.04077"},"observation_digest":"sha256:9f203769f496085f320d54520b1c59010ceca75ab6107bad1054d2fab79864c2","observation_id":"25cb378c-f7eb-4da7-8bf1-4bccb1395ec8","resolution":{"observed_at":"2026-05-20T00:00:25.356150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.08737","last_updated":"2026-05-09T06:48:00Z","snapshot_observed_at":"2026-07-06T23:20:57.084438Z","submitted_at":"2026-05-09T06:48:00Z","title":"The Extrapolation Cliff in On-Policy Distillation of Near-Deterministic Structured Outputs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-12T03:43:16.720945Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.08737"},"observation_digest":"sha256:bf32cc4e0c197dcb9b61656167317bf0990a3a16acc8c70c32792cfc7e8a7bd4","observation_id":"c4a4bfdd-8b1c-4119-86e7-800b336a3b29","resolution":{"observed_at":"2026-05-20T00:00:25.356150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.11922","last_updated":"2026-05-12T10:36:56Z","snapshot_observed_at":"2026-07-06T23:23:39.723268Z","submitted_at":"2026-05-12T10:36:56Z","title":"StepCodeReasoner: Aligning Code Reasoning with Stepwise Execution Traces via Reinforcement Learning","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-13T05:27:37.521421Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.11922"},"observation_digest":"sha256:9bc6940c3570501d76f05ec2ce37c8a896c00f58a06a21a5fe303bcae85d157b","observation_id":"f7a37feb-abee-4813-881e-d0f92d949949","resolution":{"observed_at":"2026-05-20T00:00:25.356150Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.19282","last_updated":"2026-05-19T03:00:26Z","snapshot_observed_at":"2026-07-06T23:30:02.207108Z","submitted_at":"2026-05-19T03:00:26Z","title":"Rethinking Muon Beyond Pretraining: Spectral Failures and High-Pass Remedies for VLA and RLVR","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T07:14:31.613251Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.19282"},"observation_digest":"sha256:8d1737ab9f0fc7493662060aa6ffe0cb8a6f022b8d68221826440714241a4727","observation_id":"d1d7b383-b0bd-4240-a93f-669a63b9a4a9","resolution":{"observed_at":"2026-05-20T07:18:07.221238Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.19425","last_updated":"2026-05-19T06:23:43Z","snapshot_observed_at":"2026-07-06T23:30:11.337726Z","submitted_at":"2026-05-19T06:23:43Z","title":"When to Stop Reusing: Dynamic Gradient Gating for Sample-Efficient RLVR","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T07:31:23.194325Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.19425"},"observation_digest":"sha256:3172b555c3c0eb3cfbf0262d07208114ea06c0309c98948c98e690030cd968bb","observation_id":"e34a2f17-0ed7-4d71-ba82-b47be20c4cd8","resolution":{"observed_at":"2026-05-20T07:33:07.388058Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.20865","last_updated":"2026-05-20T08:01:01Z","snapshot_observed_at":"2026-07-06T23:31:23.407780Z","submitted_at":"2026-05-20T08:01:01Z","title":"Multi-Step Likelihood-Ratio Correction for Reinforcement Learning with Verifiable Rewards","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-21T05:55:45.654673Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.20865"},"observation_digest":"sha256:fa772dbf0b7eabbd661edc2afa9b3feb7f693d29fbd66a0bf90c2a0e1a832704","observation_id":"ab02dfc7-fdc2-4d79-b05a-3c9c1dd9e6b0","resolution":{"observed_at":"2026-05-21T05:59:41.078831Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.22703","last_updated":"2026-05-21T16:45:31Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T16:45:31Z","title":"Clipping Bottleneck: Stabilizing RLVR via Stochastic Recovery of Near-Boundary Signals","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-22T07:50:44.907952Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.22703"},"observation_digest":"sha256:fa44c8f25a33d4824ccbee635a77a104a7364bb76add812959627e7c453c5a68","observation_id":"b91137a3-0dbd-4d08-a95c-7173221044fe","resolution":{"observed_at":"2026-05-22T07:51:15.747619Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2605.27846","last_updated":"2026-05-27T02:04:00Z","snapshot_observed_at":"2026-07-06T23:37:31.844094Z","submitted_at":"2026-05-27T02:04:00Z","title":"EAPO: Entropy-Driven Adaptive Positive-Negative Sample Weighting for Policy Optimization in Open-Ended QA","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-29T12:58:13.585763Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2605.27846"},"observation_digest":"sha256:88f8dede4c4b939385b502a3ad52ec85ae7ca0e5d92a8c452e37b10da6630f4b","observation_id":"3b91f462-e27a-453a-9667-af6f74dc20a1","resolution":{"observed_at":"2026-06-29T13:03:26.321365Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2606.09932","last_updated":"2026-06-07T17:58:58Z","snapshot_observed_at":"2026-07-06T23:49:12.754871Z","submitted_at":"2026-06-07T17:58:58Z","title":"When RL Fails after SFT: Rejuvenating Model Plasticity for Robust SFT-to-RL Handoff","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-06-27T18:49:47.876179Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2606.09932"},"observation_digest":"sha256:f24f97327e928d0f9a81531f321399e353bbfa061d5a4e540aad3c5c6879eae2","observation_id":"b10ce576-b196-4b3a-9133-3de48ddd63f7","resolution":{"observed_at":"2026-07-02T22:27:26.474701Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"cited_work":{"arxiv_id":"2510.06062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.06062","snapshot_observed_at":"2026-07-09T22:26:37.216917Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","venue":"cs.CL","work_id":"9d58b069-561d-440c-9c42-aba858a81448","year":2025},"citing_paper":{"arxiv_id":"2607.06987","last_updated":"2026-07-08T04:21:42Z","snapshot_observed_at":"2026-07-11T23:18:47.858943Z","submitted_at":"2026-07-08T04:21:42Z","title":"UP: Unbounded Positive Asymmetric Optimization for Breaking the Exploration-Stability Dilemma","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-09T22:18:13.418579Z"},"links":{"cited_paper":"/paper/2510.06062","citing_paper":"/paper/2607.06987"},"observation_digest":"sha256:fb1e3378213e183078e51ec5b2274458b92f75849d855b173725848c74667841","observation_id":"f62037ce-208b-412f-8689-8761caf75da3","resolution":{"observed_at":"2026-07-09T22:26:37.218159Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2510.06062/citation-record","integrity":"/paper/2510.06062/integrity","json":"/paper/2510.06062/citation-record.json","paper":"/paper/2510.06062"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.13585","last_updated":"2025-06-16T15:08:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-16T15:08:02Z","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","version":1},"cited_work":{"arxiv_id":"2506.13585","doi":"10.1109/tkde.2022.3168611","metadata_source":"pith","pith_arxiv_id":"2506.13585","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","venue":"cs.CL","work_id":"c59fbe20-f41e-4140-a81c-40a12e7e8364","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2506.13585","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:e3f53c85ef50c5381e7e10e6d37760c0b0d9906b3e7ecb23690923c3284dc745","observation_id":"c6916d2d-64c6-4933-839e-720cf355372d","resolution":{"observed_at":"2026-05-21T20:30:35.536429Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22617","last_updated":"2025-05-28T17:38:45Z","snapshot_observed_at":"2026-07-06T21:32:23.537939Z","submitted_at":"2025-05-28T17:38:45Z","title":"The Entropy Mechanism of Reinforcement Learning for Reasoning Language Models","version":1},"cited_work":{"arxiv_id":"2505.22617","doi":"10.48550/arxiv.2505.22617","metadata_source":"pith","pith_arxiv_id":"2505.22617","snapshot_observed_at":"2026-07-10T21:57:38.198014Z","title":"The Entropy Mechanism of Reinforcement Learning for Reasoning Language Models","venue":"cs.LG","work_id":"d4b4aee4-d20f-4572-886a-4ba9ea6c9b81","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2505.22617","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:f076f4765ab7733882f4b5dd2fc28e78023451edb313bf95523813f9e4c8a91f","observation_id":"837fe42b-9dff-41b4-8336-f1e5dbdef574","resolution":{"observed_at":"2026-05-21T20:30:35.543637Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:964a6a67b7414893ebbed1814661b04fb8d9bc1c80744ff6de9e341038aad071","observation_id":"43e7d9aa-e16e-4330-9fb3-75d090eccacb","resolution":{"observed_at":"2026-05-21T20:30:35.533182Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22312","last_updated":"2025-05-29T09:07:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-28T12:56:04Z","title":"Skywork Open Reasoner 1 Technical Report","version":2},"cited_work":{"arxiv_id":"2505.22312","doi":"10.48550/arxiv.2505.22312","metadata_source":"pith","pith_arxiv_id":"2505.22312","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Skywork Open Reasoner 1 Technical Report","venue":"cs.LG","work_id":"27740ebb-6704-4ef9-8878-5b81d4269869","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2505.22312","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:1890a52b6ac7c459bc5fe1a67edf7cd55a26eb9579f13065d680fae0ee6ae61d","observation_id":"d67ef9bc-3c4e-44f8-a40c-accf4cfadfc7","resolution":{"observed_at":"2026-05-21T20:30:35.290742Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0006.361316","doi":"10.1145/3600006.3613163","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention , booktitle =","venue":null,"work_id":"1b10f2a9-a178-4d23-97fb-8db2354c7e6c","year":2023},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:16dcd1018dc6b6dbad0b1ca3b19ab6aa8b947fe7d8fad0142b8361d27b203ade","observation_id":"0a934e0c-4bda-462a-ba83-401595a0c766","resolution":{"observed_at":"2026-05-21T20:30:35.300541Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"59d9582a-f946-42ad-a7f8-6e0515762556","year":2022},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:9cd1b33d7eeeed785dddeda7790281a014c83d00b0dfe3cfec38f42d40c37288","observation_id":"ade1cf8d-21a6-4a1f-a843-857ee90fe673","resolution":{"observed_at":"2026-05-21T20:30:35.696257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1126/science.abq1158","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T12:53:50.117909Z","title":"Mankowitz, Esme Sutherland Robson, Pushmeet Kohli, Nando De Freitas, Koray Kavukcuoglu, and Oriol Vinyals","venue":null,"work_id":"cc452f34-3d34-41ff-9206-8edad6625ce6","year":2022},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:9154e39aaf9f8ae38473f8fc822e3d7d69bd5ae3af7074f22f43cb37bd8cdaed","observation_id":"fcc9f190-c4b7-418f-b900-549b25fec25f","resolution":{"observed_at":"2026-05-21T20:30:35.294764Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-11T15:53:23.512078+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T15:53:23.512078+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24864","last_updated":"2025-05-30T17:59:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-30T17:59:01Z","title":"ProRL: Prolonged Reinforcement Learning Expands Reasoning Boundaries in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.24864","doi":"10.48550/arxiv.2505.24864","metadata_source":"pith","pith_arxiv_id":"2505.24864","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"ProRL: Prolonged Reinforcement Learning Expands Reasoning Boundaries in Large Language Models","venue":"cs.CL","work_id":"b6fae4a8-0f64-45dd-b9f4-22f8e0a7e7f6","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2505.24864","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:862a536193f2126a1f3f98a500746bb3cc73b2c9c4ae1b4c162a2c9671c4c434","observation_id":"9619e9b8-149d-40fc-b8af-29a8728bebe8","resolution":{"observed_at":"2026-05-21T20:30:35.530451Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06703","last_updated":"2025-02-10T17:30:23Z","snapshot_observed_at":"2026-07-06T20:34:06.899427Z","submitted_at":"2025-02-10T17:30:23Z","title":"Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling","version":1},"cited_work":{"arxiv_id":"2502.06703","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06703","snapshot_observed_at":"2026-07-04T21:00:08.398075Z","title":"Liu, R., Gao, J., Zhao, J., Zhang, K., Li, X., Qi, B., Ouyang, W., and Zhou, B","venue":null,"work_id":"9325fd2c-00cc-47d0-9154-bfa2a83c5179","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2502.06703","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:b1e86adcbe4919db7433a94779fdd8c730adc481166b61bca3d42d00fb36ee67","observation_id":"2a18a124-832f-43ac-b30b-6e04511b983b","resolution":{"observed_at":"2026-05-21T20:30:35.518062Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:4f50442159585b392724b425fd4dd67e90d2bb58758cccf9e5166f8601a95972","observation_id":"aa994ca8-bedd-40d2-be5b-6aafcb817360","resolution":{"observed_at":"2026-05-21T20:30:35.550011Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:7356760de256c85f19123c3e81a95aeca1fdd46d53c23550c94bc98e2fa3d5a3","observation_id":"e73a185f-2c59-4961-84fc-fa1df9968932","resolution":{"observed_at":"2026-05-21T20:30:35.540414Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9031.369607","doi":"10.1145/3689031.3696072","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Hybridflow: A flexible and efficient rlhf framework","venue":null,"work_id":"4909736c-3fe1-4820-b489-cca51669c6d2","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:530f3921f6951a766fb7c53e61390553b7076d396bd7e56802e5fe35d4cd64d6","observation_id":"cc404e30-0181-48d0-a3cf-5ca05a5c6ecf","resolution":{"observed_at":"2026-05-21T20:30:35.263155Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.15778","last_updated":"2026-05-15T04:36:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T16:34:01Z","title":"Stabilizing Knowledge, Promoting Reasoning: Dual-Token Constraints for RLVR","version":2},"cited_work":{"arxiv_id":"2507.15778","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.15778","snapshot_observed_at":"2026-07-04T16:19:57.768398Z","title":"Stabilizing Knowledge, Promoting Reasoning: Dual-Token Constraints for RLVR","venue":"cs.CL","work_id":"b5286fd9-b359-49d2-a6bc-e370a3293e23","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2507.15778","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:02c150ce2a1a54a818abe075bd73d22a5fdd680e6c66808858a30a00d41ca951","observation_id":"350cbf6f-eced-4230-870f-6ec2430a160f","resolution":{"observed_at":"2026-05-21T20:30:35.527450Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-07-11T03:07:50.815080Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:f01aa1232ec1a59fd9e1cee2459771e2aa9fde35f74246695abd1ed0cfca56c6","observation_id":"5f5b6ca3-da9e-4f11-bfb1-9ef25971734e","resolution":{"observed_at":"2026-05-21T20:30:35.285333Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.05118","last_updated":"2025-04-11T02:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-07T14:21:11Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","version":3},"cited_work":{"arxiv_id":"2504.05118","doi":"10.1109/access.2024.3384487","metadata_source":"pith","pith_arxiv_id":"2504.05118","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","venue":"cs.AI","work_id":"c2351652-65f7-47cd-ae80-dbcd72a6eb20","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2504.05118","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:45fdc44d1717fb2c23354b1610fb440b768000b29314f58cf142a8016414a9d6","observation_id":"aa34e41b-b028-4c4c-87b7-ebd7c5eb4527","resolution":{"observed_at":"2026-05-21T20:30:35.546640Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.18892","last_updated":"2025-08-06T08:42:32Z","snapshot_observed_at":"2026-07-06T20:57:57.039376Z","submitted_at":"2025-03-24T17:06:10Z","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","version":3},"cited_work":{"arxiv_id":"2503.18892","doi":"10.48550/arxiv.2503.18892","metadata_source":"pith","pith_arxiv_id":"2503.18892","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","venue":"cs.LG","work_id":"94a68437-02e7-425a-91b2-5846ddcbd38c","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2503.18892","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:9a690bca1508505503aadee47cedcd220b4ff29fa9c140781d99fc44f50c9757","observation_id":"c9037920-5672-4486-984e-63abb982aecd","resolution":{"observed_at":"2026-05-21T20:30:35.524502Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"cited_work":{"arxiv_id":"2509.08827","doi":"10.48550/arxiv.2509.08827","metadata_source":"pith","pith_arxiv_id":"2509.08827","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","venue":"cs.CL","work_id":"7618c14b-e527-4268-9926-d4f462ea9925","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2509.08827","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:754369b1d447de6109314a54e2c643797af8cbb0ddf31002bcbc34d8c145d344","observation_id":"35924ace-f966-4dd4-94de-0fc27fbf2332","resolution":{"observed_at":"2026-05-21T20:30:35.274924Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18071","last_updated":"2025-07-28T11:11:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-24T03:50:32Z","title":"Group Sequence Policy Optimization","version":2},"cited_work":{"arxiv_id":"2507.18071","doi":"10.48550/arxiv.2507.18071","metadata_source":"pith","pith_arxiv_id":"2507.18071","snapshot_observed_at":"2026-07-10T14:27:07.145784Z","title":"Group Sequence Policy Optimization","venue":"cs.LG","work_id":"3a98b53b-9f52-4d95-adf7-89353c0a9a65","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"cited_paper":"/paper/2507.18071","citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:b9086ff5074006b02b5f7f0c519f2039ae81666bc3aeb3eadf86670ad80a2b78","observation_id":"287e9b65-1b50-4ae0-aaf9-1d529ae3146c","resolution":{"observed_at":"2026-05-21T20:30:35.521394Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"•DAPO(Yu et al., 2025): A strong OSRL algorithm built upon GRPO (Shao et al., 2024)","venue":null,"work_id":"696d5f40-f3d7-4b57-b6b5-76438a2c6588","year":2025},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:1e53f13a39e2d9d9b8ecd3d3da1ca8114f3b31a712ff624eb91e21e25cdc83c1","observation_id":"ba978783-ff9a-4bd1-a81e-84006ad18068","resolution":{"observed_at":"2026-05-21T20:30:35.691979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"For coding, we employ DeepCoder (Luo et al., 2025a), CodeContests (Li et al., 2022), and CodeForces (Penedo et al","venue":null,"work_id":"c31bdb45-43b2-4d70-a2d1-5b6aabd9c455","year":2022},"citing_paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-21T20:29:25.874620Z"},"links":{"citing_paper":"/paper/2510.06062"},"observation_digest":"sha256:870165d3593d2f519a05548816c645facdad67e84b68b36cb863d3702bcff503","observation_id":"47938baa-b011-4cd9-b898-f17670c4a5c2","resolution":{"observed_at":"2026-05-21T20:30:35.694087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-07-20T06:30:07.809122+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2510.06062","last_updated":"2026-05-15T05:25:06Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-07T15:54:24Z","title":"When Importance Sampling Misallocates Credit: Asymmetric Ratios for Outcome-Supervised RL"},"reference_resolution":{"displayed":20,"state_counts":{"malformed_identifier":0,"metadata_mismatch":9,"parse_uncertain":0,"unresolved":0,"verified_exact":8,"verified_fuzzy":3},"total_outbound_references":20},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-07-20T06:30:07.809122+00:00","source":"crossref"},{"observed_at":"2026-07-20T06:30:01.33724+00:00","source":"retraction_watch"}],"thesis":"As of 20 July 2026, this Paper Citation Record lists 20 of 20 outbound references and 13 inbound Pith citation observations for arXiv:2510.06062."}