{"as_of":"2026-08-06T03:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3dae26e00de7477d8f508a7c28cbbe9e55cb7671ddd384f547402c3df9e5de9c","coverage":[{"denominator":22,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T11:07:18.839899Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-15T03:13:14.384567Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-15T03:14:52.571593Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2601.18150","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.18150","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","venue":"cs.LG","work_id":"92bfb8c5-79c4-4e4c-92fc-a75a829e1e82","year":2026},"citing_paper":{"arxiv_id":"2604.06916","last_updated":"2026-04-08T10:14:47Z","snapshot_observed_at":"2026-07-06T22:55:15.791334Z","submitted_at":"2026-04-08T10:14:47Z","title":"FP4 Explore, BF16 Train: Diffusion Reinforcement Learning via Efficient Rollout Scaling","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T18:10:06.994557Z"},"links":{"cited_paper":"/paper/2601.18150","citing_paper":"/paper/2604.06916"},"observation_digest":"sha256:9bf6044c7bc7bd3f1d592307e4c1a0806adb7c0a35d4ae7856467871b628aa1b","observation_id":"c5ce8129-5817-47e9-9c82-6ab934b73057","resolution":{"observed_at":"2026-05-11T05:21:08.293476Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2601.18150","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.18150","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","venue":"cs.LG","work_id":"92bfb8c5-79c4-4e4c-92fc-a75a829e1e82","year":2026},"citing_paper":{"arxiv_id":"2605.13907","last_updated":"2026-05-13T03:36:57Z","snapshot_observed_at":"2026-07-06T23:25:25.361350Z","submitted_at":"2026-05-13T03:36:57Z","title":"AIS: Adaptive Importance Sampling for Quantized RL","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-15T03:13:14.384567Z"},"links":{"cited_paper":"/paper/2601.18150","citing_paper":"/paper/2605.13907"},"observation_digest":"sha256:d9481a4436d533d9c844a6822a82a46054aa0a34896a1a2904757c0297d7852d","observation_id":"5128f4a1-330b-493a-b192-b16bd07de1ce","resolution":{"observed_at":"2026-05-15T03:14:52.575296Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2601.18150/citation-record","integrity":"/paper/2601.18150/integrity","json":"/paper/2601.18150/citation-record.json","paper":"/paper/2601.18150"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2511.14617","last_updated":"2026-04-03T12:47:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-18T16:12:21Z","title":"Seer: Online Context Learning for Fast Synchronous LLM Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2511.14617","doi":null,"metadata_source":"pith","pith_arxiv_id":"2511.14617","snapshot_observed_at":"2026-07-04T06:49:38.047014Z","title":"Seer: Online Context Learning for Fast Synchronous LLM Reinforcement Learning","venue":"cs.DC","work_id":"03b265b6-317b-4574-a145-82e7e81c521d","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"cited_paper":"/paper/2511.14617","citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:4376cf13ee5a183824d8308dbd774cab4316e7e6f516a97be3a0081812eb94d4","observation_id":"e1812e0d-54ce-4e72-aed1-0004af5a4f7f","resolution":{"observed_at":"2026-05-16T11:07:47.112397Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"TensorRT LLM","venue":null,"work_id":"2b486799-d14d-4f48-86a2-ffb61a820823","year":null},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:5e8f6048275493ab5482ef1b215599e42ba577816f7675b1c34789f6b64ea9ee","observation_id":"7819ca9a-cdc5-45ef-8686-25b09cba5c31","resolution":{"observed_at":"2026-05-16T11:10:52.906622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAtten - tion","venue":null,"work_id":"de73845a-1547-4fb2-bb0b-736fcd3eefea","year":2023},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:ddac99d9510c9e6121acac5b7333d2c9babf2ac627672bda4413dccaf85da593","observation_id":"5e2a1471-9d5d-4d12-8ef0-63f0d0af4507","resolution":{"observed_at":"2026-05-16T11:10:52.901339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Available: https://github.com/sgl-project/sglang","venue":null,"work_id":"bb925854-9ceb-4bfb-ae04-79a2646aee40","year":null},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:a0514337b376817c8f577f785551d1a1b6dd47e8f3bb8e2827b09fc54b50ffb3","observation_id":"a14b7fc8-1490-4e63-8d09-8bbc71786235","resolution":{"observed_at":"2026-05-16T11:10:52.928087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"When Speed Kills Stability: Demystifying RL Collapse from the Training-Inference Mismatch","venue":null,"work_id":"bdc6f3ab-f3a7-44e9-ab0f-05fae046e428","year":null},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:2409791c67f3e87cb578b81679678ffe416ea9ab8da6c2b1637850c879a0ccc3","observation_id":"04bf0a96-7197-468b-a166-31a5da24b390","resolution":{"observed_at":"2026-05-16T11:10:52.904224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FlashRL: 8Bit Rollouts, Full Power RL","venue":null,"work_id":"e46322cc-68f9-4b9a-8e8e-2393d23fd6cc","year":null},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:9561d653cb35a9bfdaafb498e0f0cbb5d2ad6e6bd32db299af1f5192d17cd730","observation_id":"84c9e339-d854-43ef-b0e5-388f622760f3","resolution":{"observed_at":"2026-05-16T11:10:52.923197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Your Efficient RL Framework Secretly Brings You Off-Policy RL Training","venue":null,"work_id":"02f4c25a-2c21-4dda-b436-5c008399d18e","year":null},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:a2065e75f460b13bfeb9051debd9a10e03dd9ef071ceb13290202a15fa99c850","observation_id":"29a1a3f3-4afa-45bd-87b2-12bcd3962041","resolution":{"observed_at":"2026-05-16T11:10:52.896140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DeepSeek-V3 Technical Report","venue":null,"work_id":"2d704ff0-e516-4b12-9f6b-95c24edb2ddc","year":2024},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:d3456ee6a7b8d41a0026a82df5207c0d9e0a0d06bd71259e59a3f5b487d7bf8d","observation_id":"2a4821ba-b753-4866-b0d0-e3bdcd52ba2d","resolution":{"observed_at":"2026-05-16T11:10:52.925889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19256","last_updated":"2024-10-02T04:01:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-28T06:20:03Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","version":2},"cited_work":{"arxiv_id":"2409.19256","doi":"10.1145/3689031.3696075.url:","metadata_source":"pith","pith_arxiv_id":"2409.19256","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"HybridFlow: A Flexible and Efficient RLHF Framework","venue":"cs.LG","work_id":"7eb9c9f4-b322-4bba-8011-09ff8d6ad801","year":2024},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"cited_paper":"/paper/2409.19256","citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:59aa5c5ed7df60c3d44706ade7c6068ceaac3e6c283d4b6e6e505d91ab573754","observation_id":"e8f2a045-5ed5-4ea7-afdc-2ea26f0da161","resolution":{"observed_at":"2026-05-16T11:07:47.088515Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":null,"work_id":"09bcae22-22fc-44ae-a739-48a9c04eb4f2","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:0ef49e013cbee54c96b41cfc575efe586dc066e89f4d706b6dfb0fcf0ca7ceec","observation_id":"b3f1567e-d1b4-4c92-802d-cfeda2296eae","resolution":{"observed_at":"2026-05-16T11:10:52.918601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NeMo RL: A Scalable and Efficient Post-Training Library","venue":null,"work_id":"f5d1f1b5-b47e-48ef-89ec-3cba63596e76","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:a7275ab15747cea72695eb677524b7028bc9bdbc8f64853078c5c979238c1bcc","observation_id":"11acfa0e-579d-4b2c-b2ba-1ade549b10e7","resolution":{"observed_at":"2026-05-16T11:10:52.920681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FP8 Formats for Deep Learning","venue":null,"work_id":"a08093c3-8644-48ab-bbea-494b92d3b2e1","year":2022},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:0fffa11069131c7a3b587437f8c96fa886f27821c0c4407c30e66a8fdcbb1707","observation_id":"9072d075-1b32-41f7-b6ef-a6f3ce81af98","resolution":{"observed_at":"2026-05-16T11:10:52.916167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FP8-LM: Training FP8 Large Language Models","venue":null,"work_id":"aab6d086-9bc4-4bc8-abcf-d32d25e3baa9","year":2023},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:fd99a875a07e4ce629894ef253b21016f71388ea65235a1461c541fc028424fd","observation_id":"b03e7fe6-fdf2-4310-93f7-039ae18c397e","resolution":{"observed_at":"2026-05-16T11:10:52.898697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01320","last_updated":"2023-08-02T18:49:57Z","snapshot_observed_at":"2026-07-06T16:01:45.794350Z","submitted_at":"2023-08-02T18:49:57Z","title":"DeepSpeed-Chat: Easy, Fast and Affordable RLHF Training of ChatGPT-like Models at All Scales","version":1},"cited_work":{"arxiv_id":"2308.01320","doi":"10.48550/arxiv.2308.01320","metadata_source":"arxiv_reference","pith_arxiv_id":"2308.01320","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deepspeed-chat: Easy, fast and affordable rlhf training of chatgpt-like models at all scales","venue":"arXiv (Cornell University)","work_id":"3d244678-a776-4393-8ca9-6b3c9a8e3fde","year":2023},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"cited_paper":"/paper/2308.01320","citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:f1c10c84d8a70e136a2bf9c7ab268c921d553e5a304dc4a490654520d0226c92","observation_id":"6a8f4821-a09c-4daa-a10a-b230ba7d5b2e","resolution":{"observed_at":"2026-05-16T11:07:47.104402Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"cited_work":{"arxiv_id":"2405.11143","doi":"10.48550/arxiv.2405.11143","metadata_source":"pith","pith_arxiv_id":"2405.11143","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","venue":"cs.AI","work_id":"70fa48c9-2f84-49f6-9aca-37476e021fc3","year":2024},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"cited_paper":"/paper/2405.11143","citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:9c6ed715bee02a45eeca595971a6901128a6e9f6884502d5438a6e029c81dffb","observation_id":"156a7ee1-7416-45ab-b851-7fef121056b8","resolution":{"observed_at":"2026-05-16T11:07:47.108354Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.06122","last_updated":"2025-06-06T14:33:56Z","snapshot_observed_at":"2026-07-06T21:37:56.477028Z","submitted_at":"2025-06-06T14:33:56Z","title":"Reinforcement Learning Optimization for Large-Scale Learning: An Efficient and User-Friendly Scaling Library","version":1},"cited_work":{"arxiv_id":"2506.06122","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.06122","snapshot_observed_at":"2026-07-04T10:29:45.796847Z","title":"Reinforcement learning optimization for large-scale learning: An efficient and user-friendly scaling library.arXiv preprint arXiv:2506.06122, 2025a","venue":null,"work_id":"37914127-0152-4730-8654-8b15734bd5f9","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"cited_paper":"/paper/2506.06122","citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:c4f0d97bdc5e2fbe62205ef9301805037d7df97bcbf546467a212e960ff7f302","observation_id":"a890e8d4-eb8d-409d-8cd3-39bcc6009c17","resolution":{"observed_at":"2026-05-16T11:07:47.084927Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"slime: An LLM post-training framework for RL Scaling","venue":null,"work_id":"ae36b153-37eb-44cf-8f12-be73a2ec803b","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:f6488a14d2e7c3a730aeaa788ccbb0d04e90cebd3e78e22fd79e01c5934a78de","observation_id":"cb54d84e-2a83-486e-b70a-fe5bda2cccfd","resolution":{"observed_at":"2026-05-16T11:10:52.911466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24298","last_updated":"2026-03-02T03:07:29Z","snapshot_observed_at":"2026-08-02T16:10:09.199211Z","submitted_at":"2025-05-30T07:18:25Z","title":"AReaL: A Large-Scale Asynchronous Reinforcement Learning System for Language Reasoning","version":5},"cited_work":{"arxiv_id":"2505.24298","doi":"10.48550/arxiv.2505.24298","metadata_source":"pith","pith_arxiv_id":"2505.24298","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AReaL: A Large-Scale Asynchronous Reinforcement Learning System for Language Reasoning","venue":"cs.LG","work_id":"60017c83-cbbd-4807-b8e4-f8149a6aeaf0","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"cited_paper":"/paper/2505.24298","citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:07616edf025fc80150ff93fc0bcda8dd2c5e3f94a0e2f8845ab8f8d9560d2f2b","observation_id":"26e97cb3-bbe2-4ffc-aa2d-49d925b892c4","resolution":{"observed_at":"2026-05-16T11:07:47.092261Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Small Leak Can Sink a Great Ship–Boost RL Training on MoE with IcePop!","venue":null,"work_id":"8e838eda-d901-4c9b-951b-81f1513c09a6","year":null},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:84c9a5b57bc480011184bb390759bfde5807468ffbcb40bc832472dd44f49046","observation_id":"ab328949-9340-4e9b-a113-4afae7ab48cf","resolution":{"observed_at":"2026-05-16T11:10:52.914046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.11370","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T22:26:37.153429Z","title":"Stabilizing MoE Reinforcement Learning by Aligning Training and Inference Routers","venue":null,"work_id":"4df9a4a1-3908-4d16-9c0b-253bb245a0c4","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:f0ee4a553d3e85d7c881f1166e15448f9cac612c1f57595effff7deed8a41568","observation_id":"90735c17-72d1-4819-ba3c-517e0749ae07","resolution":{"observed_at":"2026-05-16T11:07:47.100142Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"No More Train-Inference Mismatch: Bitwise Consistent On-Policy Reinforcement Learning with vLLM and TorchTitan","venue":null,"work_id":"fe39bf06-6d57-4707-add4-ecb140a4bf1e","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:6c50a7c9a4eb09d35b1ecacf522496c6b66aa73de08054c73133f008034b56ce","observation_id":"4d0bf14a-325d-4424-b52d-b4508ac02842","resolution":{"observed_at":"2026-05-16T11:10:52.908965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.26788","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T05:59:38.002040Z","title":"Defeating the training-inference mismatch via fp16","venue":null,"work_id":"0130b8e1-ddae-4ab2-b584-1de5b6918454","year":2025},"citing_paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T11:07:18.839899Z"},"links":{"citing_paper":"/paper/2601.18150"},"observation_digest":"sha256:d108e96b53ab68b824ea395681985e00e1fbe996ec51981c3786a0e3bed63245","observation_id":"bfc898b0-4975-4f2c-9624-557041151ca3","resolution":{"observed_at":"2026-05-16T11:07:47.096305Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2601.18150","last_updated":"2026-04-10T15:41:56Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-02T16:35:13.446446Z","submitted_at":"2026-01-26T05:12:05Z","title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning"},"reference_resolution":{"displayed":22,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":7,"verified_fuzzy":14},"total_outbound_references":22},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 22 of 22 outbound references and 2 inbound Pith citation observations for arXiv:2601.18150."}