{"as_of":"2026-08-08T01:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5e94f806c5c0be5646364d276fefab72812b778f49e9fe39004eb3ae7df95c66","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":11,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":11,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:34:54.497908Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T20:46:14.356541Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":"2504.04950","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-07-01T20:46:14.356541Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":"d0c04c3e-d429-46e4-b6ba-9d51a5fda758","year":2025},"citing_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-02T16:13:31.498470Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"reference_index":157,"source":"pdf_text","source_observed_at":"2026-05-11T05:26:04.960844Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2505.07062"},"observation_digest":"sha256:50e644e42dc3f0c0bded33ce7997ecf417ac7db39589a11008ebda529ccb44ff","observation_id":"6b797002-0f2c-4569-b19c-7217b5948dae","resolution":{"observed_at":"2026-05-11T05:26:05.751164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-08-07T14:34:54.497908Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization.arXiv preprint arXiv:2504.04950, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18531","last_updated":"2025-05-24T05:50:07Z","snapshot_observed_at":"2026-08-07T14:27:49.003803Z","submitted_at":"2025-05-24T05:50:07Z","title":"Generative RLHF-V: Learning Principles from Multi-modal Human Preference","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:34:54.497908Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2505.18531"},"observation_digest":"sha256:ab1091a5e1c87a2ca9cb1d6218eb9a8a76684de2cc36ef852a3fed68425ccf68","observation_id":"79be537c-ece5-4495-b294-07b595d9fd67","resolution":{"observed_at":"2026-08-07T14:34:54.497908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-08-07T13:09:00.584149Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.22648","last_updated":"2025-08-10T06:05:46Z","snapshot_observed_at":"2026-08-07T13:00:07.473210Z","submitted_at":"2025-05-28T17:57:07Z","title":"WebDancer: Towards Autonomous Information Seeking Agency","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T13:09:00.584149Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2505.22648"},"observation_digest":"sha256:771fa4d02dda80fe1d885ca47bbf6a9fc3f95ffd3b93a82ac7145780079f525d","observation_id":"8e894e89-aaa8-40d4-a55e-a065552fb817","resolution":{"observed_at":"2026-08-07T13:09:00.584149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-08-07T04:39:37.841550Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization.arXiv preprint arXiv:2504.04950, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.10406","last_updated":"2025-06-12T06:59:35Z","snapshot_observed_at":"2026-08-07T12:11:20.313302Z","submitted_at":"2025-06-12T06:59:35Z","title":"PAG: Multi-Turn Reinforced LLM Self-Correction with Policy as Generative Verifier","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T04:39:37.841550Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2506.10406"},"observation_digest":"sha256:cf5510dee2e4b0a3c08af86cf742b6b459827900951bd3a14e8a225c74ec90a0","observation_id":"972cb59f-2d79-474c-be90-22b1351106c5","resolution":{"observed_at":"2026-08-07T04:39:37.841550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-08-04T20:09:01.240055Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization.arXiv preprint arXiv:2504.04950, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.08826","last_updated":"2025-09-10T17:59:31Z","snapshot_observed_at":"2026-08-07T13:58:40.131270Z","submitted_at":"2025-09-10T17:59:31Z","title":"RewardDance: Reward Scaling in Visual Generation","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-04T20:09:01.240055Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2509.08826"},"observation_digest":"sha256:6a4bc3447d2981cc8fdaf933cfe14c9d7e9ef9ba7a92667afa3590a5df8ff7e0","observation_id":"96ff8a98-aaa1-4426-84c9-7d72a24e8711","resolution":{"observed_at":"2026-08-04T20:09:01.240055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-08-04T09:28:13.017326Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.15514","last_updated":"2026-05-24T15:31:17Z","snapshot_observed_at":"2026-08-04T09:28:11.420694Z","submitted_at":"2025-10-17T10:34:59Z","title":"Voting with the Graph: Stable RLAIF via Topological Consistency Maximization","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-04T09:28:13.017326Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2510.15514"},"observation_digest":"sha256:9fe4bda3d28ea7a802d342ea4648bd44d0bc37845f19a8d32fa86fead526a061","observation_id":"64bc43cc-a96f-4ce5-b2fb-14d01e7ae83d","resolution":{"observed_at":"2026-08-04T09:28:13.017326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":"2504.04950","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-07-01T20:46:14.356541Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":"d0c04c3e-d429-46e4-b6ba-9d51a5fda758","year":2025},"citing_paper":{"arxiv_id":"2510.24235","last_updated":"2026-04-20T06:29:42Z","snapshot_observed_at":"2026-08-03T23:17:37.743021Z","submitted_at":"2025-10-28T09:43:47Z","title":"PaTaRM: Bridging Pairwise and Pointwise Signals via Preference-Aware Task-Adaptive Reward Modeling","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T03:15:57.744706Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2510.24235"},"observation_digest":"sha256:74e43f3e7089713fb94a97bb000d484cf6a2cd593bf61de1459760007ff5c75c","observation_id":"aa743bd7-6033-40e5-8722-9cbe0250b7bd","resolution":{"observed_at":"2026-05-18T03:20:49.249595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":"2504.04950","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-07-01T20:46:14.356541Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":"d0c04c3e-d429-46e4-b6ba-9d51a5fda758","year":2025},"citing_paper":{"arxiv_id":"2604.27505","last_updated":"2026-05-20T07:08:10Z","snapshot_observed_at":"2026-07-06T23:12:57.730833Z","submitted_at":"2026-04-30T06:54:39Z","title":"Leveraging Verifier-Based Reinforcement Learning in Image Editing","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-07T08:00:33.307429Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2604.27505"},"observation_digest":"sha256:fb0f3b308eaf741bb846f96934736155ea8429427b2f6525598451deae72e7fc","observation_id":"f7d6ff37-2f61-492e-a2ec-ca2b96a8755e","resolution":{"observed_at":"2026-05-12T10:06:27.558499Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":"2504.04950","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-07-01T20:46:14.356541Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":"d0c04c3e-d429-46e4-b6ba-9d51a5fda758","year":2025},"citing_paper":{"arxiv_id":"2604.27505","last_updated":"2026-05-20T07:08:10Z","snapshot_observed_at":"2026-07-06T23:12:57.730833Z","submitted_at":"2026-04-30T06:54:39Z","title":"Leveraging Verifier-Based Reinforcement Learning in Image Editing","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-21T09:11:02.183133Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2604.27505"},"observation_digest":"sha256:4b7cc80c9424a34c42f61ae73c8a42cf1c668eba394f7d4c959cc5ae45d4c91f","observation_id":"b758234d-2223-4943-955d-1e58f95674e2","resolution":{"observed_at":"2026-05-21T09:14:05.945999Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":"2504.04950","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-07-01T20:46:14.356541Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":"d0c04c3e-d429-46e4-b6ba-9d51a5fda758","year":2025},"citing_paper":{"arxiv_id":"2605.18191","last_updated":"2026-05-18T10:31:34Z","snapshot_observed_at":"2026-08-03T01:31:13.834606Z","submitted_at":"2026-05-18T10:31:34Z","title":"Pairwise Preference Reward and Group-Based Diversity Enhancement for Superior Open-Ended Generation","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-20T10:15:22.633882Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2605.18191"},"observation_digest":"sha256:e94671a8ea20cebcd54130c09e8f8cea630fce79417686b7c48571ef0962db51","observation_id":"16782ab1-f999-4139-bbfe-0df663bba745","resolution":{"observed_at":"2026-05-20T10:18:11.891575Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","version":1},"cited_work":{"arxiv_id":"2504.04950","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.04950","snapshot_observed_at":"2026-07-01T20:46:14.356541Z","title":"A unified pairwise framework for rlhf: Bridging generative reward modeling and policy optimization","venue":null,"work_id":"d0c04c3e-d429-46e4-b6ba-9d51a5fda758","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":197,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2504.04950","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:b13bd1db8ed21cb99d3bd98cff27320332f893dda3706f419d1acce7c6ea5430","observation_id":"5104f631-dd80-4742-9611-63305d747dbf","resolution":{"observed_at":"2026-07-01T20:46:14.358178Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2504.04950/citation-record","integrity":"/paper/2504.04950/integrity","json":"/paper/2504.04950/citation-record.json","paper":"/paper/2504.04950"},"outbound":[],"paper":{"arxiv_id":"2504.04950","last_updated":"2025-04-07T11:34:48Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T16:08:01.051306Z","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 11 inbound Pith citation observations for arXiv:2504.04950."}