{"as_of":"2026-08-07T19:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6be4e22dbdda4eecd0c56cd1a2e518a208d759b82ffd413c0baf043d234c8802","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T01:04:33.188878Z","state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T12:45:56.736538Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:49:37.712152Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"cited_work":{"arxiv_id":"2506.12350","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.12350","snapshot_observed_at":"2026-07-04T06:49:37.712152Z","title":null,"venue":null,"work_id":"e6f1a23e-33e3-431b-a8ab-e51d7cdd6f75","year":2025},"citing_paper":{"arxiv_id":"2606.02340","last_updated":"2026-06-01T14:48:19Z","snapshot_observed_at":"2026-08-02T12:22:54.067388Z","submitted_at":"2026-06-01T14:48:19Z","title":"Transitivity in Inhomogeneous Random Tournaments","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-28T12:45:56.736538Z"},"links":{"cited_paper":"/paper/2506.12350","citing_paper":"/paper/2606.02340"},"observation_digest":"sha256:125be26d195d4012743a6b8e9ab19c128e0e1fbfbf8e2e690e41ce269afc7d7b","observation_id":"381d7141-7902-4833-9a51-9c632f59e13b","resolution":{"observed_at":"2026-07-02T01:06:24.252873Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"cited_work":{"arxiv_id":"2506.12350","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.12350","snapshot_observed_at":"2026-07-04T06:49:37.712152Z","title":null,"venue":null,"work_id":"e6f1a23e-33e3-431b-a8ab-e51d7cdd6f75","year":2025},"citing_paper":{"arxiv_id":"2606.21550","last_updated":"2026-06-19T15:47:01Z","snapshot_observed_at":"2026-08-02T16:31:22.749179Z","submitted_at":"2026-06-19T15:47:01Z","title":"AI Alignment From Social Choice Perspectives","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-26T14:12:36.892697Z"},"links":{"cited_paper":"/paper/2506.12350","citing_paper":"/paper/2606.21550"},"observation_digest":"sha256:94bab592c3595682b5dd69732ae5485f08439e2ce3e29207b94300877572e59c","observation_id":"c78663c3-2f78-44a3-b449-3e9eaf0d3262","resolution":{"observed_at":"2026-07-04T06:49:37.714149Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.12350/citation-record","integrity":"/paper/2506.12350/integrity","json":"/paper/2506.12350/citation-record.json","paper":"/paper/2506.12350"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.710047Z","title":"Therefore, a preference matching distribution exists","venue":null,"work_id":"3ffd90a4-3e50-4372-a9d9-0bef2f47381b","year":2023},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.970029Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:347be8908e22ce3f175d0ff2ec1526752ad78736d0729d2e84a8053656d563c9","observation_id":"5b11d9d1-86ef-42c9-8247-4415029ad08c","resolution":{"observed_at":"2026-08-07T01:04:34.815076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.10271","last_updated":"2024-06-04T14:34:38Z","snapshot_observed_at":"2026-07-06T18:00:44.208590Z","submitted_at":"2024-04-16T03:59:33Z","title":"Social Choice Should Guide AI Alignment in Dealing with Diverse Human Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.10271","snapshot_observed_at":"2026-08-07T01:04:30.630057Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.630057Z"},"links":{"cited_paper":"/paper/2404.10271","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:a0819f71836f08e8dcba585de5ec7bd8c4c69e366f2dd211d228254ea0eafab1","observation_id":"d6efb15f-2b56-4a38-b874-2a29f44bad6b","resolution":{"observed_at":"2026-08-07T01:04:30.630057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13038","last_updated":"2024-04-19T17:49:56Z","snapshot_observed_at":"2026-07-06T18:02:49.215786Z","submitted_at":"2024-04-19T17:49:56Z","title":"Mapping Social Choice Theory to RLHF","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13038","snapshot_observed_at":"2026-08-07T01:04:30.717742Z","title":"Mapping social choice theory to RLHF.arXiv preprint arXiv:2404.13038,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.717742Z"},"links":{"cited_paper":"/paper/2404.13038","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:89ac7de8b5a214bdff6f548864e9470d9fb25d22c13a2104ecc7280b26a53ef3","observation_id":"a701d99e-19f1-4c20-b602-d115cabc1f85","resolution":{"observed_at":"2026-08-07T01:04:30.717742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10584","last_updated":"2024-02-25T19:15:26Z","snapshot_observed_at":"2026-07-06T17:04:09.326801Z","submitted_at":"2023-12-17T02:14:15Z","title":"Policy Optimization in RLHF: The Impact of Out-of-preference Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10584","snapshot_observed_at":"2026-08-07T01:04:31.048119Z","title":"Policy optimization in rlhf: The impact of out-of-preference data","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.048119Z"},"links":{"cited_paper":"/paper/2312.10584","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:4dbd56b41563f82e1bb6a62208c8d1b8235f36fa487436dd981c326f9ac187e5","observation_id":"37cfa10b-53f2-4d10-807e-fffd6a485e01","resolution":{"observed_at":"2026-08-07T01:04:31.048119Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19266","last_updated":"2025-01-31T16:26:28Z","snapshot_observed_at":"2026-07-06T20:29:07.258163Z","submitted_at":"2025-01-31T16:26:28Z","title":"Jackpot! Alignment as a Maximal Lottery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.19266","snapshot_observed_at":"2026-08-07T01:04:31.269934Z","title":"Jackpot! alignment as a maximal lottery.arXiv preprint arXiv:2501.19266,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.269934Z"},"links":{"cited_paper":"/paper/2501.19266","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:ef435c8193ab7a04b3eecede4b447021f87e176690d928c9f44d57adfe661c07","observation_id":"e1530569-83dd-4e7b-9891-ac9968612789","resolution":{"observed_at":"2026-08-07T01:04:31.269934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16048","last_updated":"2023-10-24T17:59:04Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:59:04Z","title":"AI Alignment and Social Choice: Fundamental Limitations and Policy Implications","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16048","snapshot_observed_at":"2026-08-07T01:04:31.412637Z","title":"AI alignment and social choice: Fundamental limitations and policy implications","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.412637Z"},"links":{"cited_paper":"/paper/2310.16048","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:1d48ec3f854dea937f942f045f98c346ed7067869cafc9da9ccc1d71b192b66a","observation_id":"f056459f-c02d-4a49-91fb-13dadd438e7c","resolution":{"observed_at":"2026-08-07T01:04:31.412637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00886","last_updated":"2024-06-11T16:25:52Z","snapshot_observed_at":"2026-07-06T16:55:53.666800Z","submitted_at":"2023-12-01T19:26:23Z","title":"Nash Learning from Human Feedback","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00886","snapshot_observed_at":"2026-08-07T01:04:31.531656Z","title":"Nash learning from human feedback.arXiv preprint arXiv:2312.00886,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.531656Z"},"links":{"cited_paper":"/paper/2312.00886","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:856a16bfb4d688af18057358e1b840091bf912cc5ebf6f4d335f985eb3ee82ed","observation_id":"af2407b8-2d5e-41c8-a244-5ddb9e26359a","resolution":{"observed_at":"2026-08-07T01:04:31.531656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00254","last_updated":"2024-05-27T14:08:40Z","snapshot_observed_at":"2026-08-07T08:11:05.128316Z","submitted_at":"2024-04-30T23:57:23Z","title":"RLHF from Heterogeneous Feedback via Personalization and Preference Aggregation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00254","snapshot_observed_at":"2026-08-07T01:04:31.735488Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.735488Z"},"links":{"cited_paper":"/paper/2405.00254","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:83cf1962b0596f51e61d091a890249b39670a41f5db62b412ec4645b5a9637e7","observation_id":"88259934-527d-46dc-b40b-0a6c150812d2","resolution":{"observed_at":"2026-08-07T01:04:31.735488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-07T01:04:31.830315Z","title":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.830315Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:5d1f18eae103be7d49359773cabdfe5e43aea495580c22b2024fa4747b95a7ea","observation_id":"d60b0dca-4b57-46e9-82f0-746fb3ec9b80","resolution":{"observed_at":"2026-08-07T01:04:31.830315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08358","last_updated":"2024-04-17T01:58:09Z","snapshot_observed_at":"2026-07-06T17:01:13.998131Z","submitted_at":"2023-12-13T18:51:34Z","title":"Distributional Preference Learning: Understanding and Accounting for Hidden Context in RLHF","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08358","snapshot_observed_at":"2026-08-07T01:04:32.028901Z","title":"Distributional preference learn- ing: Understanding and accounting for hidden context in RLHF.arXiv preprint arXiv:2312.08358,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.028901Z"},"links":{"cited_paper":"/paper/2312.08358","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:81b5e02a3b19aa379471576344cff47c5a937554b6a9d59dc08aa8401811ed28","observation_id":"699e2fc7-0744-4fe0-9bf4-189127a6bc72","resolution":{"observed_at":"2026-08-07T01:04:32.028901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08448","last_updated":"2024-05-14T09:12:30Z","snapshot_observed_at":"2026-08-05T00:57:12.421643Z","submitted_at":"2024-05-14T09:12:30Z","title":"Understanding the performance gap between online and offline alignment algorithms","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08448","snapshot_observed_at":"2026-08-07T01:04:32.093625Z","title":"Understanding the perfor- mance gap between online and offline alignment algorithms.arXiv preprint arXiv:2405.08448,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.093625Z"},"links":{"cited_paper":"/paper/2405.08448","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:ff75a697720be719f73fd1176de317a8208819e2f4930c6bd4a69eb696b5cb32","observation_id":"92b86db4-ffbe-4594-be5f-5f1bfa313a7d","resolution":{"observed_at":"2026-08-07T01:04:32.093625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T01:04:32.208926Z","title":"Gemini: a family of highly capable multimodal models.arXiv preprint arXiv:2312.11805,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.208926Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:363638e92acb63eebf275151f7fe0392fdec388388510f7b7e956c207e2dc280","observation_id":"f2767bef-1860-470f-b3ee-a9578e512d76","resolution":{"observed_at":"2026-08-07T01:04:32.208926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T01:04:32.286520Z","title":"Llama: Open and efficient foundation language models.arXiv preprint arXiv:2302.13971,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.286520Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:2d46642a5d3afedc0efbf365c340f18b592ced6cd4ef53dc08dab2e4372a9d53","observation_id":"e894c17b-8779-4d88-b5c3-7ae66709f369","resolution":{"observed_at":"2026-08-07T01:04:32.286520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:35.460658Z","title":"Zhilin Wang, Yi Dong, Jiaqi Zeng, Virginia Adams, Makesh Narsimhan Sreedhar, Daniel Egert, Olivier Delalleau, Jane Scowcroft, Neel Kant, Aidan Swope, et al","venue":null,"work_id":"0921af18-79d1-4a8f-bc7b-480d66ed1fb1","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.350165Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:ef4ebe21a75426bee8d723f31bc80eefafaac25d9d314ac8f5fd1d54a8e87a43","observation_id":"c8b38568-b696-4339-b24e-92ad0dda7007","resolution":{"observed_at":"2026-08-07T01:04:35.593726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16455","last_updated":"2025-08-25T01:01:38Z","snapshot_observed_at":"2026-07-06T18:20:01.382394Z","submitted_at":"2024-05-26T07:00:05Z","title":"On the Algorithmic Bias of Aligning Large Language Models with RLHF: Preference Collapse and Matching Regularization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16455","snapshot_observed_at":"2026-08-07T01:04:32.454213Z","title":"On the algorithmic bias of aligning large language models with rlhf: Preference collapse and matching regularization.arXiv preprint arXiv:2405.16455,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.454213Z"},"links":{"cited_paper":"/paper/2405.16455","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:d2e390a61c467ee1ce56775462a15a0f50ccf2aeaa0d235d3354d223dee25eaa","observation_id":"5f3c4032-9f7c-458e-964b-d3b0f333e6ba","resolution":{"observed_at":"2026-08-07T01:04:32.454213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:32.522528Z","title":"Restoring calibration for aligned large language models: A calibration-aware fine-tuning approach","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.522528Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:ba35fa2b176bc6e610485825238c5a5a08d833495b0f1aa7ffa35453a0cd8c09","observation_id":"7a2cbe65-6812-4d08-8b31-c3678f0730a1","resolution":{"observed_at":"2026-08-07T01:04:32.522528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:35.287569Z","title":"Iterative preference learning from human feedback: Bridging theory and practice for rlhf under kl- constraint","venue":null,"work_id":"b1af7fe7-1996-490c-b64b-1a07293b26e7","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.636118Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:497bbb5d98d948ecb574600f68712909c31f3f343e626481cce889aa9237d9bf","observation_id":"7defe16f-f701-4713-bf0f-45a0706a7933","resolution":{"observed_at":"2026-08-07T01:04:35.373688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:35.113189Z","title":"Asymptotics of language model alignment","venue":null,"work_id":"32c3c79d-9fd3-4947-9f23-4e138b965e06","year":2027},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.741501Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:e82d895cbd1014477ea37d331de00b7e211b56b90f7ce30937924c328157b5b2","observation_id":"b33ced5a-b656-45c1-8310-f2c9ba8bd1ee","resolution":{"observed_at":"2026-08-07T01:04:35.202878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05006","last_updated":"2024-03-08T03:05:11Z","snapshot_observed_at":"2026-07-06T17:41:23.460300Z","submitted_at":"2024-03-08T03:05:11Z","title":"Provable Multi-Party Reinforcement Learning with Diverse Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05006","snapshot_observed_at":"2026-08-07T01:04:32.828481Z","title":"Provable multi-party reinforcement learning with diverse human feedback.arXiv preprint arXiv:2403.05006,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.828481Z"},"links":{"cited_paper":"/paper/2403.05006","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:df936f1a2ea319d781155fcb9ece59810251d8353a9df100d5e1e1d02814fa3c","observation_id":"3fcb66fe-0fc2-42c9-a6cf-fd8b5072d6a7","resolution":{"observed_at":"2026-08-07T01:04:32.828481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.904498Z","title":null,"venue":null,"work_id":"4a957b9e-0011-4f1d-868f-c1120fe9688d","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:32.907603Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:ed89ef091f3473d22d691edc580ace6a7ee6b353ba7bdd7e2d70aa08031e62ed","observation_id":"c87aacea-1141-4d61-b508-6f6cba18e980","resolution":{"observed_at":"2026-08-07T01:04:35.004266Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.504628Z","title":"The work by Tang et al","venue":null,"work_id":"27f8739e-91b8-426d-aca0-6976ab5065d1","year":2017},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:33.030370Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:259e3bc022071784fb4f6cf550cb9068fbffa429ae25488e2665fbf5f77df638","observation_id":"6ff25476-08d1-443b-8c00-679588b028d9","resolution":{"observed_at":"2026-08-07T01:04:34.587683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.250344Z","title":"Outside of fine-tuning, model editing has emerged as a complementary strategy to modify LLM behavior across tasks [Jin et al., 2025]","venue":null,"work_id":"7470cf3e-ba0d-4db5-9b56-d2e700c84d52","year":2025},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:33.117798Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:fba0d3dadc72dca6ec41ab807c43b5c33d5096b82bd211d06c488c368607a239","observation_id":"861998dd-ec03-4cda-bb5b-b6c69066a382","resolution":{"observed_at":"2026-08-07T01:04:34.387321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T01:04:34.078966Z","title":"[2024], which constrain its robustness relative to reinforcement learning techniques like PPO","venue":null,"work_id":"12c5a699-5442-47d6-9b87-325f6a2bf491","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:33.188878Z"},"links":{"citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:74a33913e2d147f4efcbeba5b2d94ee08198acf9fc02c3a953e483ef11110ed5","observation_id":"7aa67290-0996-43c2-8634-c003f6d7a217","resolution":{"observed_at":"2026-08-07T01:04:34.151963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09656","last_updated":"2025-02-25T10:19:35Z","snapshot_observed_at":"2026-07-06T18:00:16.713949Z","submitted_at":"2024-04-15T10:44:31Z","title":"Learn Your Reference Model for Real Good Alignment","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09656","snapshot_observed_at":"2026-08-07T01:04:30.926721Z","title":"Learn your reference model for real good alignment.arXiv preprint arXiv:2404.09656,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2006,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.926721Z"},"links":{"cited_paper":"/paper/2404.09656","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:6529483f93268dffd77eb3f9b19485fa383c3ff598c3ac903717900fb97f0e3e","observation_id":"cad5854e-c103-4d4f-8d73-ebc7127bc9aa","resolution":{"observed_at":"2026-08-07T01:04:30.926721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02612","last_updated":"2024-06-19T17:08:13Z","snapshot_observed_at":"2026-07-06T18:09:46.934515Z","submitted_at":"2024-05-04T08:43:45Z","title":"Learning Linear Utility Functions From Pairwise Comparison Queries","version":3},"cited_work":{"arxiv_id":"2405.02612","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.02612","snapshot_observed_at":"2026-08-07T01:04:33.833571Z","title":"Learning Linear Utility Functions From Pairwise Comparison Queries","venue":"cs.LG","work_id":"a2bcb8f1-deb1-4c52-b209-82ffca95ca0c","year":2024},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2009,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.810972Z"},"links":{"cited_paper":"/paper/2405.02612","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:3f9017637fe8493304d0f91b2995658d493234e684836fec928636ab5a652657","observation_id":"1b4f020e-eabd-446e-9266-f8a868b6dc13","resolution":{"observed_at":"2026-08-07T01:04:33.883031Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-03T04:49:15.195814Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-07T01:04:30.242850Z","title":"Sparks of artificial general intelligence: Early experiments with gpt-4.arXiv preprint arXiv:2303.12712,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.242850Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:1939a7fe0be26ebaef61d7267e4d46af0ae973c38ded1859cdced1fa4091626a","observation_id":"80dbad57-95b5-4a4b-a3b3-b85a5bc8976e","resolution":{"observed_at":"2026-08-07T01:04:30.242850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20627","last_updated":"2025-05-27T02:07:35Z","snapshot_observed_at":"2026-08-07T13:48:08.093334Z","submitted_at":"2025-05-27T02:07:35Z","title":"Fundamental Limits of Game-Theoretic LLM Alignment: Smith Consistency and Preference Matching","version":1},"cited_work":{"arxiv_id":"2505.20627","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20627","snapshot_observed_at":"2026-08-07T01:04:33.548624Z","title":"Fundamental Limits of Game-Theoretic LLM Alignment: Smith Consistency and Preference Matching","venue":"cs.GT","work_id":"2aac02d0-c8b6-4d01-9b77-d4f27398331c","year":2025},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.958892Z"},"links":{"cited_paper":"/paper/2505.20627","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:2c978efee01e449f8eb0e638d89d06ae051e81e23b2504e1bce4f70ba608bb39","observation_id":"5f332d57-5d8f-4179-ba1e-cbc868b75180","resolution":{"observed_at":"2026-08-07T01:04:33.572321Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T01:04:31.655495Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.655495Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:f4c3ce55eeaee2d120f34e922d519d94ba4500d6819dad0fb5069454a18f1394","observation_id":"0093612e-68fb-4bcc-999e-531b37b21540","resolution":{"observed_at":"2026-08-07T01:04:31.655495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08925","last_updated":"2024-12-26T00:15:20Z","snapshot_observed_at":"2026-07-06T17:29:51.807428Z","submitted_at":"2024-02-14T03:56:27Z","title":"MaxMin-RLHF: Alignment with Diverse Human Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08925","snapshot_observed_at":"2026-08-07T01:04:30.349050Z","title":"URLhttps://openreview.net/forum?id=bx24KpJ4Eb","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.349050Z"},"links":{"cited_paper":"/paper/2402.08925","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:e5344592239941473decc370e696e7239fac61ee5d9d1be6bf951abcfff30f94","observation_id":"074dd878-1664-4c9d-88dc-fa80b3d653da","resolution":{"observed_at":"2026-08-07T01:04:30.349050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08495","last_updated":"2024-04-16T17:36:39Z","snapshot_observed_at":"2026-07-06T17:59:21.208430Z","submitted_at":"2024-04-12T14:25:49Z","title":"Dataset Reset Policy Optimization for RLHF","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08495","snapshot_observed_at":"2026-08-07T01:04:30.492218Z","title":"Dataset reset policy optimization for rlhf.arXiv preprint arXiv:2404.08495,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:30.492218Z"},"links":{"cited_paper":"/paper/2404.08495","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:b0882f463ee374cffc5a550cb3c67cd7a3107998986f4e088800be5705f532dd","observation_id":"19322493-ba98-49d0-bc1a-954f3af1adf5","resolution":{"observed_at":"2026-08-07T01:04:30.492218Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10990","last_updated":"2026-04-30T19:06:42Z","snapshot_observed_at":"2026-07-06T20:52:27.848221Z","submitted_at":"2025-03-14T01:29:21Z","title":"Statistical Impossibility and Possibility of Aligning LLMs with Human Preferences: From Condorcet Paradox to Nash Equilibrium","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10990","snapshot_observed_at":"2026-08-07T01:04:31.156730Z","title":"Kaizhao Liu, Qi Long, Zhekun Shi, Weijie J Su, and Jiancong Xiao","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T01:04:31.156730Z"},"links":{"cited_paper":"/paper/2503.10990","citing_paper":"/paper/2506.12350"},"observation_digest":"sha256:e13b17041983c267b98fdb302c1a472bf2f60bb2b2b1cd2c932a7dfc559598ab","observation_id":"8a24506e-5f96-4b25-8720-621acdde4d0f","resolution":{"observed_at":"2026-08-07T01:04:31.156730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.12350","last_updated":"2025-06-14T05:14:49Z","latest_version":1,"primary_category":"stat.ML","snapshot_observed_at":"2026-08-07T00:49:43.707111Z","submitted_at":"2025-06-14T05:14:49Z","title":"Theoretical Tensions in RLHF: Reconciling Empirical Success with Inconsistencies in Social Choice Theory"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":7},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 2 inbound Pith citation observations for arXiv:2506.12350."}