{"as_of":"2026-08-01T19:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:daa5702c96a93aa5d8cc427c835b08e2357f683609f03f9b4d125fee06f8ee12","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":9,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":9,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-01T06:32:01.292127+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T20:01:06.623978Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-03T17:58:47.896297Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2604.16995","last_updated":"2026-04-18T13:49:47Z","snapshot_observed_at":"2026-07-06T23:04:10.324443Z","submitted_at":"2026-04-18T13:49:47Z","title":"SPS: Steering Probability Squeezing for Better Exploration in Reinforcement Learning for Large Language Models","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-10T06:57:03.100519Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2604.16995"},"observation_digest":"sha256:3fe0d92491651528a1d04c6cce4fad81c837030ee5beee794bac2f6e3e3d98fd","observation_id":"964beff2-08cf-4683-916e-3735a1a8ba7c","resolution":{"observed_at":"2026-05-15T01:43:11.007781Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2605.06139","last_updated":"2026-05-20T08:43:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-07T12:38:17Z","title":"Listwise Policy Optimization: Group-based RLVR as Target-Projection on the LLM Response Simplex","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-08T13:55:22.422923Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2605.06139"},"observation_digest":"sha256:136719705492d0e9cfe8c41eada472db33447dec924ee4af97aa630fe3875487","observation_id":"da0e14bb-d075-4a25-bd41-f43183387ad6","resolution":{"observed_at":"2026-05-15T01:43:11.007781Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2605.06139","last_updated":"2026-05-20T08:43:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-07T12:38:17Z","title":"Listwise Policy Optimization: Group-based RLVR as Target-Projection on the LLM Response Simplex","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-21T09:02:28.407064Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2605.06139"},"observation_digest":"sha256:abacce19dd56e62a262854047bbff841a66b9820a60d4ab66fc5457359f9c970","observation_id":"d8860af1-1bc7-47e7-81df-981c76b9dd47","resolution":{"observed_at":"2026-05-21T09:04:04.472275Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2605.08327","last_updated":"2026-05-08T17:00:38Z","snapshot_observed_at":"2026-07-06T23:20:33.816373Z","submitted_at":"2026-05-08T17:00:38Z","title":"Interactive Critique-Revision Training for Reliable Structured LLM Generation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T00:58:20.234462Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2605.08327"},"observation_digest":"sha256:57561f61a1f88de2666841201693ae290ae41a7141810433085aa54a5006440b","observation_id":"aac73811-644f-48b5-93ff-571c0e86ac05","resolution":{"observed_at":"2026-05-15T01:43:11.007781Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2605.21266","last_updated":"2026-05-20T14:53:43Z","snapshot_observed_at":"2026-07-06T23:31:46.877766Z","submitted_at":"2026-05-20T14:53:43Z","title":"How Much Online RL is Enough? Informative Rollouts for Offline Preference Optimization in RLVR","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-21T06:22:55.286281Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2605.21266"},"observation_digest":"sha256:2fb3baa64c3b2a4f5d2bc7c4760163db855960d58206794faf84590550e4e555","observation_id":"bb338d90-c674-4863-ad86-9e49946b8157","resolution":{"observed_at":"2026-05-21T06:24:00.558406Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2606.09371","last_updated":"2026-06-08T11:48:55Z","snapshot_observed_at":"2026-07-06T23:48:44.159537Z","submitted_at":"2026-06-08T11:48:55Z","title":"Capability-Aligned Hierarchical Learning for Tool-Augmented LLMs","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-27T16:43:00.259139Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2606.09371"},"observation_digest":"sha256:b76b5e08df6aedc70dd1eba8259b4841a773d167d9d5599f86df51353dc29e81","observation_id":"9068efe5-88a0-4fcb-b288-3f4441d23828","resolution":{"observed_at":"2026-07-03T01:17:30.710581Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2606.17250","last_updated":"2026-06-15T19:49:42Z","snapshot_observed_at":"2026-08-01T19:14:09.872274Z","submitted_at":"2026-06-15T19:49:42Z","title":"Rethinking Groups in Critic-Free RLVR","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-06-27T03:22:14.586529Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2606.17250"},"observation_digest":"sha256:4fdcac6bf914a991e4a90375fd30cc3b032ee16f99fddc3231604f326140f493","observation_id":"42e87114-b9c6-46d0-be4a-b409ff485a96","resolution":{"observed_at":"2026-07-03T17:58:47.897450Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":"2510.00977","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-03T17:58:47.896297Z","title":"It Takes Two: Your GRPO Is Secretly DPO","venue":"cs.LG","work_id":"15d941cc-60bc-41eb-b65d-68ee8f196b92","year":2025},"citing_paper":{"arxiv_id":"2607.00152","last_updated":"2026-06-30T20:28:08Z","snapshot_observed_at":"2026-07-07T00:05:49.557137Z","submitted_at":"2026-06-30T20:28:08Z","title":"GRPO, Dr. GRPO, and DAPO Are Three Operations on One Number: The Group-Standard-Deviation Identity","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-02T19:52:43.147416Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2607.00152"},"observation_digest":"sha256:cc6c44c19a80870f3fa556dfbffe714954a35f45f5c6e15b6323a6fd72bd7090","observation_id":"467aa4b5-4c6a-4ae4-927f-8a362a877226","resolution":{"observed_at":"2026-07-02T19:57:19.054101Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.00977","snapshot_observed_at":"2026-07-11T20:01:06.623978Z","title":"Wu et al.,It Takes Two: Your GRPO Is Secretly DPO, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.04332","last_updated":"2026-07-05T14:29:18Z","snapshot_observed_at":"2026-07-11T20:01:05.655213Z","submitted_at":"2026-07-05T14:29:18Z","title":"On the effectiveness of reward functions in reinforcement learning for confidence calibration of large language models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T20:01:06.623978Z"},"links":{"cited_paper":"/paper/2510.00977","citing_paper":"/paper/2607.04332"},"observation_digest":"sha256:272e8a73ce36dfa380a4fe0f4a8b1b075433e963c4fabefe7c99fe9973de9b98","observation_id":"8cdab405-7cf5-4d94-be43-5d3540f04d9a","resolution":{"observed_at":"2026-07-11T20:01:06.623978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2510.00977/citation-record","integrity":"/paper/2510.00977/integrity","json":"/paper/2510.00977/citation-record.json","paper":"/paper/2510.00977"},"outbound":[],"paper":{"arxiv_id":"2510.00977","last_updated":"2026-05-14T04:25:16Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T22:31:24.978726Z","submitted_at":"2025-10-01T14:52:11Z","title":"It Takes Two: Your GRPO Is Secretly DPO"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"thesis":"As of 1 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 9 inbound Pith citation observations for arXiv:2510.00977."}