{"as_of":"2026-08-08T17:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:207a0d1b784d9c4906741f8f132d989cce230e53b0fae0148f169bb74946e09b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T15:03:47.464904Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T11:09:46.406911Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2411.10442","last_updated":"2025-04-07T09:09:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-15T18:59:27Z","title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T09:16:17.150383Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2411.10442"},"observation_digest":"sha256:10d81148c88e6d18fd69a954021d31f361538aa466bfce6e50d2293335042884","observation_id":"2a42b6b5-f3ad-4eb1-96b3-e897de916829","resolution":{"observed_at":"2026-05-16T09:16:17.252104Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2502.06387","last_updated":"2026-04-07T06:33:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-10T12:15:27Z","title":"How Humans Help LLMs: Assessing and Incentivizing Human Preference Annotators","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-23T03:56:18.703995Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2502.06387"},"observation_digest":"sha256:5a82094ea555451b01e5a372fe1e0c0ae42fe01db486da82f65faf25960a27b9","observation_id":"0c4164c7-dd8d-4467-a398-af8bbeda1ebf","resolution":{"observed_at":"2026-05-23T03:57:29.679899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T18:23:49.795434Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.10391","last_updated":"2025-02-14T18:59:51Z","snapshot_observed_at":"2026-08-08T01:24:46.892879Z","submitted_at":"2025-02-14T18:59:51Z","title":"MM-RLHF: The Next Step Forward in Multimodal LLM Alignment","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T18:23:49.795434Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2502.10391"},"observation_digest":"sha256:308bddf30d7e4eaa59414690ff03748898fccfd780e224f5f2fc41998336f821","observation_id":"6e9ef6e6-ad4d-4ddd-ac35-445bbf735d62","resolution":{"observed_at":"2026-08-07T18:23:49.795434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T15:19:08.819245Z","title":"R., Kini, A., and Natarajan, N","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.15694","last_updated":"2025-05-21T16:07:47Z","snapshot_observed_at":"2026-08-08T05:51:01.468645Z","submitted_at":"2025-05-21T16:07:47Z","title":"A Unified Theoretical Analysis of Private and Robust Offline Alignment: from RLHF to DPO","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:19:08.819245Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.15694"},"observation_digest":"sha256:ab0e09409b40c3ee02aa2ab6b88a614f45921773ed384f39482a8a0e962c1f8f","observation_id":"6f6217b9-2b21-4366-b350-b0d52c2f1c40","resolution":{"observed_at":"2026-08-07T15:19:08.819245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2505.19134","last_updated":"2026-04-13T23:58:15Z","snapshot_observed_at":"2026-07-06T21:30:04.828669Z","submitted_at":"2025-05-25T13:11:55Z","title":"Incentivizing High-Quality Human Annotations with Golden Questions","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T13:41:26.730528Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.19134"},"observation_digest":"sha256:13141f57b8107c82129f2c5d8c8fb1924929286c0ffc4f9d4e05d01a43a1c583","observation_id":"55493028-8900-4233-ab07-3174d5ea4748","resolution":{"observed_at":"2026-05-19T13:42:19.320740Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T13:44:56.673138Z","title":"R., Kini, A., and Natarajan, N","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21395","last_updated":"2025-05-27T16:23:24Z","snapshot_observed_at":"2026-08-07T13:26:21.678395Z","submitted_at":"2025-05-27T16:23:24Z","title":"Square$\\chi$PO: Differentially Private and Robust $\\chi^2$-Preference Optimization in Offline Direct Alignment","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T13:44:56.673138Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.21395"},"observation_digest":"sha256:1730db6e1eebc87514dcfc52ae1c5d3086a354c2323429da79c32f0c7a637c8a","observation_id":"c07e7b14-0763-4b65-88e6-5d97b0cb27d7","resolution":{"observed_at":"2026-08-07T13:44:56.673138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-07T12:27:53.423556Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24709","last_updated":"2025-05-30T15:30:43Z","snapshot_observed_at":"2026-08-07T12:12:53.183544Z","submitted_at":"2025-05-30T15:30:43Z","title":"On Symmetric Losses for Robust Policy Optimization with Noisy Preferences","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:27:53.423556Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2505.24709"},"observation_digest":"sha256:1d545c70842837ab2a889151dcd9a6d44623f3e58bd2136ef2f8edc2880a95c0","observation_id":"18383f47-ebb8-49be-82e0-db1deea0be98","resolution":{"observed_at":"2026-08-07T12:27:53.423556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-06T19:59:31.844690Z","title":"Provably robust dpo: Aligning language models with noisy feedback.arXiv preprint arXiv:2403.00409, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04136","last_updated":"2026-07-04T19:39:07Z","snapshot_observed_at":"2026-08-08T15:50:06.172618Z","submitted_at":"2025-07-05T19:13:00Z","title":"A Technical Survey of Reinforcement Learning Techniques for Large Language Models","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T19:59:31.844690Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2507.04136"},"observation_digest":"sha256:cd61e8d2c5d525a6169801ed428c1dadcf221460c38ca864b6548be9aa8f54cb","observation_id":"e72950c1-3456-4bd4-8789-0c0fb597fb4a","resolution":{"observed_at":"2026-08-06T19:59:31.844690Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2509.08933","last_updated":"2026-05-21T17:37:36Z","snapshot_observed_at":"2026-07-06T22:28:28.274132Z","submitted_at":"2025-09-10T18:56:39Z","title":"Corruption-Tolerant Asynchronous Q-Learning with Near-Optimal Rates","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T12:58:26.626172Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2509.08933"},"observation_digest":"sha256:b158cde2e0b4b40b791a66c97f61091cabf32c0f2798036bc29bfe3d792147d5","observation_id":"879ddf9d-ee39-4e6d-bd23-4b8d4c93c977","resolution":{"observed_at":"2026-05-22T13:01:34.180564Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-04T11:27:22.570045Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.05342","last_updated":"2026-06-01T10:47:12Z","snapshot_observed_at":"2026-08-06T09:31:11.235195Z","submitted_at":"2025-10-06T20:09:37Z","title":"Margin Adaptive DPO: Leveraging Reward Model for Granular Control in Preference Optimization","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T11:27:22.570045Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2510.05342"},"observation_digest":"sha256:0354b78ef2ce2730518177c1f841cf104ca892ddb7ae143272ed800a0ecafff8","observation_id":"a0f62a07-09a5-43fe-8a7f-b05f5f17f77f","resolution":{"observed_at":"2026-08-04T11:27:22.570045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2510.13830","last_updated":"2026-05-09T20:37:45Z","snapshot_observed_at":"2026-08-08T06:12:13.417479Z","submitted_at":"2025-10-10T08:57:34Z","title":"Users as Annotators: LLM Preference Learning from Comparison Mode","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T08:19:58.093621Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2510.13830"},"observation_digest":"sha256:999d6c218c44a8b89458d5f509d8573441f2f36ce84dfa82a470a3e163c90d8b","observation_id":"b1db7cef-8d31-405d-91c7-233f62db05e9","resolution":{"observed_at":"2026-05-18T08:21:06.992814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.02971","last_updated":"2026-05-07T19:25:15Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-05-03T14:22:49Z","title":"Multilingual Safety Alignment via Self-Distillation","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-08T19:35:56.059362Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.02971"},"observation_digest":"sha256:fbde637369e1041fdd3460dd2bf50e1abfeea6220a535e9f3aa2e39a8e63f47f","observation_id":"1f275270-dd03-4b92-9031-d2d2159fd45e","resolution":{"observed_at":"2026-05-09T05:45:22.410770Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.02971","last_updated":"2026-05-07T19:25:15Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-05-03T14:22:49Z","title":"Multilingual Safety Alignment via Self-Distillation","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-11T01:08:32.264867Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.02971"},"observation_digest":"sha256:97bc301fdfa1ff66e869bf2d0bfe8e58e4c56e2fb7bd4667140c73b3e268e08f","observation_id":"10a79da6-4279-4b68-9ab8-028dec443f34","resolution":{"observed_at":"2026-05-11T04:41:00.754016Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.11134","last_updated":"2026-05-29T17:16:57Z","snapshot_observed_at":"2026-08-01T16:30:31.998321Z","submitted_at":"2026-05-11T18:41:12Z","title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-13T06:30:51.812541Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.11134"},"observation_digest":"sha256:fbbc422382dc580551ad0b64513a97a09168e11551bbf2899f42fbcefe008d42","observation_id":"59f00d15-ee8a-4cfe-8867-7658461bc1a0","resolution":{"observed_at":"2026-05-13T06:32:24.273715Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2605.23398","last_updated":"2026-05-22T09:11:20Z","snapshot_observed_at":"2026-08-02T13:58:02.707128Z","submitted_at":"2026-05-22T09:11:20Z","title":"TPMM-DPO: Trajectory-aware Preference-guided Model Merging for Iterative Direct Preference Optimization","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-25T03:41:52.859647Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2605.23398"},"observation_digest":"sha256:89f6f45fc2d789da80b44a2646ce3486d53a166ff65f1b917448da5788539e1e","observation_id":"a29d9f3b-da26-431f-b3c0-00d7387e32e4","resolution":{"observed_at":"2026-05-25T03:45:17.586731Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2606.19607","last_updated":"2026-06-17T21:19:01Z","snapshot_observed_at":"2026-08-02T22:14:26.237081Z","submitted_at":"2026-06-17T21:19:01Z","title":"Which Pairs to Compare for LLM Post-Training?","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-26T20:37:19.221848Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2606.19607"},"observation_digest":"sha256:88ed5437641900bfd33d32ba4812ccc14557f9031a74db2c75092fec5dfe8efd","observation_id":"66a041cc-61ca-4ecb-90bd-fc72154bdac8","resolution":{"observed_at":"2026-07-04T01:09:19.254368Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":"2403.00409","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-04T11:09:46.406911Z","title":"Provably robust dpo: Aligning language models with noisy feed- back","venue":null,"work_id":"79a01a14-9a55-47f3-85eb-38feebf8582b","year":2024},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":1},"reference_index":190,"source":"pdf_text","source_observed_at":"2026-06-26T08:09:57.542558Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:21e35a920ceba01f1efee32ca49ae8b9f4c1d181009ad5c3ae4069ff0040b4fa","observation_id":"0d736ac7-41f0-4a83-9703-22e57c8659ea","resolution":{"observed_at":"2026-07-04T11:09:46.408648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-02T10:27:18.410065Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback.arXiv Preprint arXiv:2403.00409, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.24937","last_updated":"2026-07-27T15:17:17Z","snapshot_observed_at":"2026-08-02T23:19:25.465662Z","submitted_at":"2026-06-22T17:48:54Z","title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","version":2},"reference_index":190,"source":"pdf_text","source_observed_at":"2026-08-02T10:27:18.410065Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2606.24937"},"observation_digest":"sha256:4526a5f90c1eef95fe482cbc40b1683a4ffc7e8748a4e683b0d0b29ea8654684","observation_id":"23266883-97f4-4d2f-8f72-348a70fd816d","resolution":{"observed_at":"2026-08-02T10:27:18.410065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-07-14T15:37:01.391649Z","title":"Provably robust dpo: Aligning language models with noisy feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09796","last_updated":"2026-07-20T03:41:16Z","snapshot_observed_at":"2026-08-08T03:21:32.168317Z","submitted_at":"2026-07-09T09:20:25Z","title":"Metadata-Free Meta-Reweighted Direct Preference Optimization under Noisy Preference Labels","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-14T15:37:01.391649Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2607.09796"},"observation_digest":"sha256:d507a05f911a37afb0e5daa2d4aa474e326e50dd831b182783ec98ae64b6b131","observation_id":"f903dc6c-5203-4d51-8355-c19cb17ee593","resolution":{"observed_at":"2026-07-14T15:37:01.391649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-02T08:00:53.937134Z","title":"Provably robust dpo: Aligning language models with noisy feedback,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09796","last_updated":"2026-07-20T03:41:16Z","snapshot_observed_at":"2026-08-08T03:21:32.168317Z","submitted_at":"2026-07-09T09:20:25Z","title":"Metadata-Free Meta-Reweighted Direct Preference Optimization under Noisy Preference Labels","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T08:00:53.937134Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2607.09796"},"observation_digest":"sha256:e3fa8ed25c5b166e08b8c1ef351639bea55646e55679195f3331da6bc122b798","observation_id":"1b17ee32-4086-4ec1-9545-b8151cf549fd","resolution":{"observed_at":"2026-08-02T08:00:53.937134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00409","snapshot_observed_at":"2026-08-08T15:03:47.464904Z","title":"arXiv preprint arXiv:2403.00409 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05341","last_updated":"2026-08-05T18:59:23Z","snapshot_observed_at":"2026-08-08T17:16:12.330299Z","submitted_at":"2026-08-05T18:59:23Z","title":"Positive-Unlabeled Preference Optimization For Chest X-ray Report Generation","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T15:03:47.464904Z"},"links":{"cited_paper":"/paper/2403.00409","citing_paper":"/paper/2608.05341"},"observation_digest":"sha256:41c6ca15078852334b7be9788f584cffc0989f71e74db7528f3f97a4c9f495f0","observation_id":"bff2af43-1386-4c93-8032-bc0d540df63a","resolution":{"observed_at":"2026-08-08T15:03:47.464904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.00409/citation-record","integrity":"/paper/2403.00409/integrity","json":"/paper/2403.00409/citation-record.json","paper":"/paper/2403.00409"},"outbound":[],"paper":{"arxiv_id":"2403.00409","last_updated":"2024-04-12T01:09:37Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-04T15:58:27.584901Z","submitted_at":"2024-03-01T09:55:18Z","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2403.00409."}