{"as_of":"2026-08-08T02:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:14e29e1ba273f89170756503c841ab0473c1bcb6420289768c46c364be99d994","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:07.442164Z","state":"measured"},{"denominator":58,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":58,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T02:16:42.324571Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T21:00:39.084405Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"cited_work":{"arxiv_id":"2506.02592","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02592","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"& Lin, Y","venue":null,"work_id":"18a87cfe-1136-4565-b4ac-b4b03cb488a1","year":2025},"citing_paper":{"arxiv_id":"2509.26464","last_updated":"2026-05-19T17:20:14Z","snapshot_observed_at":"2026-07-06T22:31:15.349221Z","submitted_at":"2025-09-30T16:13:56Z","title":"Extreme Self-Preference in Language Models","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-21T20:57:37.199128Z"},"links":{"cited_paper":"/paper/2506.02592","citing_paper":"/paper/2509.26464"},"observation_digest":"sha256:f659d025b74c94451ec7452ef2dc87e7f340f25a974184606d066a5fb508949c","observation_id":"01dc12c8-47bc-4c49-b235-192caa7839ca","resolution":{"observed_at":"2026-05-21T21:00:39.086277Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"cited_work":{"arxiv_id":"2506.02592","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02592","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"& Lin, Y","venue":null,"work_id":"18a87cfe-1136-4565-b4ac-b4b03cb488a1","year":2025},"citing_paper":{"arxiv_id":"2510.07517","last_updated":"2026-04-09T17:03:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-08T20:29:46Z","title":"When Identity Skews Debate: Anonymization for Bias-Reduced Multi-Agent Reasoning","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T08:50:53.486588Z"},"links":{"cited_paper":"/paper/2506.02592","citing_paper":"/paper/2510.07517"},"observation_digest":"sha256:bc9bfa3fee19cda3e7df6d75d603e64a329a4d96b1ea995ac5fa98ae86b1a053","observation_id":"6ad59c7e-f8f6-4c67-bea4-dd910d61e0f8","resolution":{"observed_at":"2026-05-18T08:51:08.567479Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02592","snapshot_observed_at":"2026-08-04T02:16:42.324571Z","title":"Beyond the surface: Measuring self-preference in LLM judgments.arXiv preprint arXiv:2506.02592, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00017","last_updated":"2026-06-29T12:20:14Z","snapshot_observed_at":"2026-08-07T15:03:02.082641Z","submitted_at":"2026-06-29T12:20:14Z","title":"Memory Reward Inflation in Self-Improving LLM Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T02:16:42.324571Z"},"links":{"cited_paper":"/paper/2506.02592","citing_paper":"/paper/2608.00017"},"observation_digest":"sha256:0ed239b2e9d93dff62b6cacf5e5059b8acb9d27006c17a18a72002e76b7d75a9","observation_id":"a4352c5a-0e5b-440b-af5c-2a8cb2ba9922","resolution":{"observed_at":"2026-08-04T02:16:42.324571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02592/citation-record","integrity":"/paper/2506.02592/integrity","json":"/paper/2506.02592/citation-record.json","paper":"/paper/2506.02592"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:01.399959Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.399959Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:5c109e2a414bec2d04374486f16e8a5fbeef65979e271a17b0c14230b16166a9","observation_id":"1aab9281-65d6-486d-8ba9-3327a25e6110","resolution":{"observed_at":"2026-08-07T11:26:01.399959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:01.465840Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.465840Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:952594c38756e3b04161c3c14ecc49cf06506b16b6083eb13b2b93a2268883b1","observation_id":"916ecf34-c6b3-4acd-b948-635d662e7df4","resolution":{"observed_at":"2026-08-07T11:26:01.465840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:09.222283Z","title":null,"venue":null,"work_id":"95784d9d-9728-4546-90cc-b63135156667","year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.557460Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:27dd59470b6faa450d32ea256201a2ead13207a8b37687cb404b8642df509ad9","observation_id":"3024f0e2-9035-4cba-ad29-7d3a29d9bcb5","resolution":{"observed_at":"2026-08-07T11:26:09.309397Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-07T11:26:01.632516Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.632516Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:c2d5a1485ab581b993c084ab780dc994f0ab840060b3f09e78ed4acc9222d39d","observation_id":"fcc45e8a-b65a-47b3-949a-abdeb19eecd6","resolution":{"observed_at":"2026-08-07T11:26:01.632516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:01.729009Z","title":null,"venue":null,"work_id":null,"year":1952},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.729009Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:ae32c136f27725cc7db07861033540df043609a8f8d1ad18da7ddcad95e9f44f","observation_id":"e93e67a4-bb96-4958-b89f-6ac226594aba","resolution":{"observed_at":"2026-08-07T11:26:01.729009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:01.796164Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.796164Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:849a5230ccea2f3cc51dfb2d0608fe20a12995be5b47acb1d5b91c69ecfc451e","observation_id":"a6ef2e73-9a5d-487e-ac02-bcb0f2e3ec14","resolution":{"observed_at":"2026-08-07T11:26:01.796164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10669","last_updated":"2024-09-26T03:16:52Z","snapshot_observed_at":"2026-07-06T17:31:08.762955Z","submitted_at":"2024-02-16T13:21:06Z","title":"Humans or LLMs as the Judge? A Study on Judgement Biases","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10669","snapshot_observed_at":"2026-08-07T11:26:01.891765Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.891765Z"},"links":{"cited_paper":"/paper/2402.10669","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:5af3d334d9c4a4e0983eddc1cddb01eabc8d62a76cfd2074bf7fff47b7b2944a","observation_id":"2a704e88-0a57-490d-aa55-a99138c00bb1","resolution":{"observed_at":"2026-08-07T11:26:01.891765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:01.992851Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:01.992851Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:51d530ab48b817f6fe42ae26c6985d11fc6a8f88151026f79a40a15a15ec8b8c","observation_id":"44e04843-4344-4b95-9591-919fa420439a","resolution":{"observed_at":"2026-08-07T11:26:01.992851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:09.064528Z","title":null,"venue":null,"work_id":"50d10851-c6ac-486a-a803-dac3db275a01","year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:02.089341Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:1a97a539a58240eb3fdfe6ad45f82579938ec03cd732ace0f20f82a22b62ea33","observation_id":"bfd26f89-e201-463e-94c4-0b7f99aec333","resolution":{"observed_at":"2026-08-07T11:26:09.130400Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01377","last_updated":"2024-07-16T03:24:39Z","snapshot_observed_at":"2026-08-02T07:46:40.319683Z","submitted_at":"2023-10-02T17:40:01Z","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01377","snapshot_observed_at":"2026-08-07T11:26:02.214831Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:02.214831Z"},"links":{"cited_paper":"/paper/2310.01377","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:ab37245d44d498e0588b29d81ee0ea7020dd35c425c1d8587b7db9c0eeaf9b84","observation_id":"df786909-59ae-4eb3-a216-77e507584322","resolution":{"observed_at":"2026-08-07T11:26:02.214831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14233","last_updated":"2023-05-23T16:49:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T16:49:14Z","title":"Enhancing Chat Language Models by Scaling High-quality Instructional Conversations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14233","snapshot_observed_at":"2026-08-07T11:26:02.375895Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:02.375895Z"},"links":{"cited_paper":"/paper/2305.14233","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:b555dc94d2f93b322c0862440ab75abfeba77baf4f9bc0b1795498e19bfe2afd","observation_id":"511d3411-70d5-4f68-85cd-17437e3cfd96","resolution":{"observed_at":"2026-08-07T11:26:02.375895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:02.502570Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:02.502570Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:ce1d58e810aebba170bb6cbdfd24ba51901c0b9374b534558f79f312112abf1c","observation_id":"91f45ef9-147b-48e1-9b2f-0a96f1876e88","resolution":{"observed_at":"2026-08-07T11:26:02.502570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:08.837756Z","title":null,"venue":null,"work_id":"20a1b181-96b3-4f86-8e32-31a46fbe007b","year":2019},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:02.638673Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:d16ca6d0cc2cbbf26acfe402d9fbb59b1556704810411e8f564083c3ccf1b8b5","observation_id":"9031e36a-c68a-45b5-8448-46253566c8a1","resolution":{"observed_at":"2026-08-07T11:26:08.966253Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-07T11:26:02.787662Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:02.787662Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:4c41b8ee71409c63859ca9c89322a5a39348e0cc44904fece25598900f695e92","observation_id":"8d0dee2f-f657-4628-8a3d-c5f7d77129dd","resolution":{"observed_at":"2026-08-07T11:26:02.787662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T11:26:02.942801Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:02.942801Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:7e46cac6f57cc8ec4f4f8f108a1cdb80da6c86ca7afe088dfce7119884d0c7bb","observation_id":"6611b524-41ca-4621-94f6-bd82bdd9031d","resolution":{"observed_at":"2026-08-07T11:26:02.942801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-07T11:26:03.056652Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.056652Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:0912ae2a728ae6c4020a0542205131455b7bcc96e935a7639fbb2af922ba0dc4","observation_id":"2849b9fb-d3d2-4036-9aa9-29a6a1d57b8f","resolution":{"observed_at":"2026-08-07T11:26:03.056652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19085","last_updated":"2024-10-11T08:21:50Z","snapshot_observed_at":"2026-07-31T16:18:46.037116Z","submitted_at":"2024-02-29T12:12:30Z","title":"Controllable Preference Optimization: Toward Controllable Multi-Objective Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19085","snapshot_observed_at":"2026-08-07T11:26:03.173700Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.173700Z"},"links":{"cited_paper":"/paper/2402.19085","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:e85b130805adcde70b3db1de60ae5cc5dfc303ebbc6a3a7e078e95ac41d608b9","observation_id":"99aba7be-d3b8-49e9-a7fd-c8ef588971ae","resolution":{"observed_at":"2026-08-07T11:26:03.173700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T11:26:03.351926Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.351926Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:638d28c9cecdb007c15ba9464562f92f2deec4422b69714ee05174494e6fbe16","observation_id":"d8c09241-ecda-4029-b8cc-3e301a904ac1","resolution":{"observed_at":"2026-08-07T11:26:03.351926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:08.684812Z","title":null,"venue":null,"work_id":"f049afa8-bc88-40a6-a672-d3505e127990","year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.467505Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:3220ef7f5d01f2dda1e039ceb74e59d7c555f3629e43bf97ce8727d48f6924f5","observation_id":"6fb0e45e-169b-42a5-9970-6a79fd6a0456","resolution":{"observed_at":"2026-08-07T11:26:08.754298Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:26:03.625941Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.625941Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:166dd2d00047466ba646e9aba49a2d279cb5a909873f64fd975f3a46e7fc3544","observation_id":"568c9826-03d8-4c44-94b8-b56c6042850c","resolution":{"observed_at":"2026-08-07T11:26:03.625941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-07T11:26:03.732146Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.732146Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:4e6bbe0ffc203818e6f50be9ff840a007d50b76861f405b7ec2b0ae5fd538686","observation_id":"fa632138-9ef8-49ee-a22b-2b698d4f6bf8","resolution":{"observed_at":"2026-08-07T11:26:03.732146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17012","last_updated":"2024-09-25T16:57:20Z","snapshot_observed_at":"2026-07-06T16:25:21.571679Z","submitted_at":"2023-09-29T06:53:10Z","title":"Benchmarking Cognitive Biases in Large Language Models as Evaluators","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17012","snapshot_observed_at":"2026-08-07T11:26:03.842968Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.842968Z"},"links":{"cited_paper":"/paper/2309.17012","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:da55cf44052ce3f419e80d8eeb85690269a2f828969e7022fcdde24fd33b848e","observation_id":"2f4d8d5a-931a-4bb6-81bc-60471a111347","resolution":{"observed_at":"2026-08-07T11:26:03.842968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13787","last_updated":"2024-06-08T16:40:12Z","snapshot_observed_at":"2026-08-02T18:11:57.036767Z","submitted_at":"2024-03-20T17:49:54Z","title":"RewardBench: Evaluating Reward Models for Language Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13787","snapshot_observed_at":"2026-08-07T11:26:03.953001Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:03.953001Z"},"links":{"cited_paper":"/paper/2403.13787","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:439e06846a26105f490683b998fd8fa29e57bfec82ed4d583fe20763c03a64cd","observation_id":"e36f4cae-64be-45db-9f0f-1e7b3cfff12a","resolution":{"observed_at":"2026-08-07T11:26:03.953001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00267","last_updated":"2024-09-03T14:01:54Z","snapshot_observed_at":"2026-07-06T16:13:07.384791Z","submitted_at":"2023-09-01T05:53:33Z","title":"RLAIF vs. RLHF: Scaling Reinforcement Learning from Human Feedback with AI Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00267","snapshot_observed_at":"2026-08-07T11:26:04.092617Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.092617Z"},"links":{"cited_paper":"/paper/2309.00267","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:39761881187b338fc1e761a3f62c81291d7af9a6d59e133239d88d9f2587fb9a","observation_id":"a2a151bb-abec-4435-91ae-e2868aaa889b","resolution":{"observed_at":"2026-08-07T11:26:04.092617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:04.199628Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.199628Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:697b4ee3e0ed04528e8226f3e85269acd190c47732e412c2f070b8bbe58b8c80","observation_id":"9193983d-d82f-418e-b27d-2ebefab0ee75","resolution":{"observed_at":"2026-08-07T11:26:04.199628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:04.279674Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.279674Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:d1cf58a68cc343ac0b0ddee69e48de85d9a83d9dec83da278c5d7cf35ab844e2","observation_id":"fe2a4650-f233-4da7-a0b6-7a32f078144c","resolution":{"observed_at":"2026-08-07T11:26:04.279674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:04.363339Z","title":"Hashimoto","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.363339Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:2f6f039bc3b40110fac7d8280e0f404704e017cee88cb91dcecc5a4795e1a36f","observation_id":"b64f8235-083e-4ee9-b0c7-f7b01fb26ca2","resolution":{"observed_at":"2026-08-07T11:26:04.363339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.01552","last_updated":"2023-12-04T00:46:11Z","snapshot_observed_at":"2026-08-07T00:42:49.627314Z","submitted_at":"2023-12-04T00:46:11Z","title":"The Unlocking Spell on Base LLMs: Rethinking Alignment via In-Context Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.01552","snapshot_observed_at":"2026-08-07T11:26:04.503845Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.503845Z"},"links":{"cited_paper":"/paper/2312.01552","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:78443e601280956ed6620d9c38ede7e96b0cf094c93fee11ec41b1703071295c","observation_id":"3469f3d8-c65e-4e37-bb2e-5e0afe358758","resolution":{"observed_at":"2026-08-07T11:26:04.503845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.07958","last_updated":"2022-05-08T02:43:02Z","snapshot_observed_at":"2026-08-03T16:26:48.747700Z","submitted_at":"2021-09-08T17:15:27Z","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.07958","snapshot_observed_at":"2026-08-07T11:26:04.592298Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.592298Z"},"links":{"cited_paper":"/paper/2109.07958","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:31e9c21cf99430c48c33292ff63aef8d30a28355ee9aeabeadda51a9b4710298","observation_id":"59248575-72ef-42be-9662-9c1db09f8e83","resolution":{"observed_at":"2026-08-07T11:26:04.592298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T11:26:04.730166Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.730166Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:4040a071753680c2f3765d6cc2f294929da4528064314104e5df87bf7cf9cca8","observation_id":"42b79dbf-0220-4ead-866c-e13a1984be0e","resolution":{"observed_at":"2026-08-07T11:26:04.730166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18743","last_updated":"2024-08-25T09:58:57Z","snapshot_observed_at":"2026-08-06T10:19:25.039692Z","submitted_at":"2023-11-30T17:41:30Z","title":"AlignBench: Benchmarking Chinese Alignment of Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18743","snapshot_observed_at":"2026-08-07T11:26:04.837491Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.837491Z"},"links":{"cited_paper":"/paper/2311.18743","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:cc5f28c7e8f50cfb144aba52e953eab3a90c6ec6348c950c3a0d676e5ad531d9","observation_id":"df62893e-d5f8-4916-a8f8-3c71b13c46a2","resolution":{"observed_at":"2026-08-07T11:26:04.837491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09766","last_updated":"2024-06-07T09:41:36Z","snapshot_observed_at":"2026-08-04T10:35:44.012115Z","submitted_at":"2023-11-16T10:43:26Z","title":"LLMs as Narcissistic Evaluators: When Ego Inflates Evaluation Scores","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.09766","snapshot_observed_at":"2026-08-07T11:26:04.946179Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:04.946179Z"},"links":{"cited_paper":"/paper/2311.09766","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:e5577ea0951cc64a8dd15bf88b93047f8ccfbc85f64a8bf6b43885fcf09cdb2e","observation_id":"c5a216dd-79c7-4ed1-89f9-9c84f203cf38","resolution":{"observed_at":"2026-08-07T11:26:04.946179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.02295","last_updated":"2019-04-04T01:18:56Z","snapshot_observed_at":"2026-07-06T07:43:46.589620Z","submitted_at":"2019-04-04T01:18:56Z","title":"Evaluating Style Transfer for Text","version":1},"cited_work":{"arxiv_id":"1904.02295","doi":null,"metadata_source":"pith","pith_arxiv_id":"1904.02295","snapshot_observed_at":"2026-08-07T11:26:07.912486Z","title":"Evaluating Style Transfer for Text","venue":"cs.CL","work_id":"94bc02de-2c51-4682-847d-2d6de0651124","year":2019},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.025643Z"},"links":{"cited_paper":"/paper/1904.02295","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:d904f89427363cb1ab0a47ece42fcda5f0116ffe989cf79ed8aea89077df5a53","observation_id":"2652d21b-ea36-4892-a7e1-769f3ea96b18","resolution":{"observed_at":"2026-08-07T11:26:07.976111Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13577","last_updated":"2023-09-23T06:05:22Z","snapshot_observed_at":"2026-07-06T16:10:31.087739Z","submitted_at":"2023-08-25T13:07:33Z","title":"Text Style Transfer Evaluation Using Large Language Models","version":2},"cited_work":{"arxiv_id":"2308.13577","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.13577","snapshot_observed_at":"2026-08-07T11:26:07.780779Z","title":"Text Style Transfer Evaluation Using Large Language Models","venue":"cs.CL","work_id":"c486fa6a-8555-4415-934c-553c03cac790","year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.130234Z"},"links":{"cited_paper":"/paper/2308.13577","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:c3c5a51612171c0b3e1a00ea568b27536a7c53da33a575360e5fa32f70216e23","observation_id":"250c8b4a-aeea-47ce-bf7a-78d8d52a83fd","resolution":{"observed_at":"2026-08-07T11:26:07.833592Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:08.497367Z","title":null,"venue":null,"work_id":"9b2e61e7-467c-4d9f-8010-262c73e88f91","year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.237736Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:4ab8409d19f192cccf284cf435713d8056520e00de904123a6ca0a1aff5d8241","observation_id":"846299db-4fcb-47a2-9645-7711fc940f2d","resolution":{"observed_at":"2026-08-07T11:26:08.558455Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:05.347061Z","title":null,"venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.347061Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:b7d7e9109ebfa3e9677361d832ee61abd0644cb207b121e43587651f64fbb7e5","observation_id":"be9bacc0-b406-498a-a6ab-cd665dc04a99","resolution":{"observed_at":"2026-08-07T11:26:05.347061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16789","last_updated":"2023-10-03T14:45:48Z","snapshot_observed_at":"2026-07-06T16:00:46.542753Z","submitted_at":"2023-07-31T15:56:53Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16789","snapshot_observed_at":"2026-08-07T11:26:05.458126Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.458126Z"},"links":{"cited_paper":"/paper/2307.16789","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:95b720605a8114165782b494a1b05d4256ab323bd800a3b0629227cc7807eed2","observation_id":"e2bc21e0-b692-4fb5-8259-61f7dd50d2a5","resolution":{"observed_at":"2026-08-07T11:26:05.458126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:05.547451Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.547451Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:e4ea0faa69893f07f6d0b09089867429ba6ce0543415f459279d6aea0840c66e","observation_id":"6d918236-3f33-4060-8d64-fac47a29f405","resolution":{"observed_at":"2026-08-07T11:26:05.547451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05910","last_updated":"2024-04-09T23:21:45Z","snapshot_observed_at":"2026-07-06T16:30:00.114521Z","submitted_at":"2023-10-09T17:56:53Z","title":"SALMON: Self-Alignment with Instructable Reward Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05910","snapshot_observed_at":"2026-08-07T11:26:05.656399Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.656399Z"},"links":{"cited_paper":"/paper/2310.05910","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:a5107bbe2d7e2a2b743c4768a4507a5b919b4883122c0416e603f62f297a46d4","observation_id":"7b650efc-f0e8-4153-9842-6765752cf9c5","resolution":{"observed_at":"2026-08-07T11:26:05.656399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T11:26:05.795023Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.795023Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:40a63161442395e85ffda4c0c8fd623e9780ef7421e30e65b6c074378aa6a13e","observation_id":"f7d88fb0-a0cb-4aa9-b9e8-82264ccdc5cf","resolution":{"observed_at":"2026-08-07T11:26:05.795023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-07T11:26:05.913103Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:05.913103Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:04db164acf77a8189d4577f0472591f89dc19f6d682f20cee9d542687ab1ee79","observation_id":"d7a6a6c4-1e8f-4520-9ee9-fd9191237295","resolution":{"observed_at":"2026-08-07T11:26:05.913103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:06.015237Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.015237Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:45ff203e8884da03a3ab04b4acb15aa435b28b1328776d4fe4bb166f96df4240","observation_id":"3f66f203-92c3-4d48-bd0f-961683027ee7","resolution":{"observed_at":"2026-08-07T11:26:06.015237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:06.158196Z","title":null,"venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.158196Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:c95e129b7cc19af56b53864950687af67fc22f02117aefbea8a9fd7757506d6f","observation_id":"e79cbff3-326f-42d4-b14f-0f10c3b4e1bd","resolution":{"observed_at":"2026-08-07T11:26:06.158196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17926","last_updated":"2023-08-30T13:22:35Z","snapshot_observed_at":"2026-08-03T19:35:11.838629Z","submitted_at":"2023-05-29T07:41:03Z","title":"Large Language Models are not Fair Evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17926","snapshot_observed_at":"2026-08-07T11:26:06.235975Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.235975Z"},"links":{"cited_paper":"/paper/2305.17926","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:0df98a555c248fe7551f1321b5433b5235a4c3754f51a2421db19af72b40151c","observation_id":"962f9a8b-8e50-429b-93dd-db8482425c0d","resolution":{"observed_at":"2026-08-07T11:26:06.235975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05087","last_updated":"2024-05-24T06:37:31Z","snapshot_observed_at":"2026-08-07T12:09:31.277517Z","submitted_at":"2023-06-08T10:41:56Z","title":"PandaLM: An Automatic Evaluation Benchmark for LLM Instruction Tuning Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05087","snapshot_observed_at":"2026-08-07T11:26:06.346963Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.346963Z"},"links":{"cited_paper":"/paper/2306.05087","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:ebebc26a024df93bdd4f4219e6f9fc8bf21c5a1a0e3fe3f013032c867b1ca587","observation_id":"ce379c9d-1747-46e9-9e0e-bbdd090ed7be","resolution":{"observed_at":"2026-08-07T11:26:06.346963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13387","last_updated":"2023-09-04T01:47:30Z","snapshot_observed_at":"2026-07-06T16:10:23.328130Z","submitted_at":"2023-08-25T14:02:12Z","title":"Do-Not-Answer: A Dataset for Evaluating Safeguards in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.13387","snapshot_observed_at":"2026-08-07T11:26:06.467953Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.467953Z"},"links":{"cited_paper":"/paper/2308.13387","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:58679436cdef6e143e240cc40ea29e1af232e6fd203a8da2739c422cd12a21b1","observation_id":"2d3a899b-48da-4516-b36b-31962dfbab41","resolution":{"observed_at":"2026-08-07T11:26:06.467953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21819","last_updated":"2025-06-21T08:39:06Z","snapshot_observed_at":"2026-08-04T18:30:37.623897Z","submitted_at":"2024-10-29T07:42:18Z","title":"Self-Preference Bias in LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21819","snapshot_observed_at":"2026-08-07T11:26:06.576445Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.576445Z"},"links":{"cited_paper":"/paper/2410.21819","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:f5515aee84c40e93ea7d52256b19d578abcfacfc6465b9da48225b8b660e0668","observation_id":"ed677fea-c2ee-4a76-8c26-83b2f27c212e","resolution":{"observed_at":"2026-08-07T11:26:06.576445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.19594","last_updated":"2024-07-30T01:38:06Z","snapshot_observed_at":"2026-07-06T18:53:00.965448Z","submitted_at":"2024-07-28T21:58:28Z","title":"Meta-Rewarding Language Models: Self-Improving Alignment with LLM-as-a-Meta-Judge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.19594","snapshot_observed_at":"2026-08-07T11:26:06.652545Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.652545Z"},"links":{"cited_paper":"/paper/2407.19594","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:1d69ff186f4ed783f4b5a4c5f4a66ebbdaa26f66be4e32a3076b590fc9b3dac4","observation_id":"e66676d5-37b4-4900-9a0c-5c695b23be66","resolution":{"observed_at":"2026-08-07T11:26:06.652545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05692","last_updated":"2025-01-14T05:39:40Z","snapshot_observed_at":"2026-07-06T17:57:19.117933Z","submitted_at":"2024-04-08T17:18:04Z","title":"Evaluating Mathematical Reasoning Beyond Accuracy","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05692","snapshot_observed_at":"2026-08-07T11:26:06.792641Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.792641Z"},"links":{"cited_paper":"/paper/2404.05692","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:910c025b08e585c9617ad8fedef2f4cfb3912e4383e506afe54682769205b3f1","observation_id":"ac304e2e-cc51-4890-80b2-814a8191dc1f","resolution":{"observed_at":"2026-08-07T11:26:06.792641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11436","last_updated":"2024-06-18T04:41:07Z","snapshot_observed_at":"2026-07-06T17:31:41.687799Z","submitted_at":"2024-02-18T03:10:39Z","title":"Pride and Prejudice: LLM Amplifies Self-Bias in Self-Refinement","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11436","snapshot_observed_at":"2026-08-07T11:26:06.902686Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.902686Z"},"links":{"cited_paper":"/paper/2402.11436","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:8c64561a07815ec7b6c287538470de94e7f9f0e0bb1759f9f20efff9a6a09024","observation_id":"7987081c-db41-4372-a1f6-16cf1e2027c9","resolution":{"observed_at":"2026-08-07T11:26:06.902686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T11:26:06.980884Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:06.980884Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:d353b3aaaa5529600fcf60d851d75ace1b103f86697c28b20fd6410ea29123db","observation_id":"8e1b10f2-16c8-40f6-ab06-84fa8b35661b","resolution":{"observed_at":"2026-08-07T11:26:06.980884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02736","last_updated":"2024-10-04T03:57:47Z","snapshot_observed_at":"2026-08-01T08:21:19.528254Z","submitted_at":"2024-10-03T17:53:30Z","title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02736","snapshot_observed_at":"2026-08-07T11:26:07.054479Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:07.054479Z"},"links":{"cited_paper":"/paper/2410.02736","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:085d6e2f4658d564260de7184209dedf872c71d83fc3e27c72b83d81103265be","observation_id":"63155177-c41f-45d0-9a61-73881ae447db","resolution":{"observed_at":"2026-08-07T11:26:07.054479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.19669","last_updated":"2024-10-14T12:19:44Z","snapshot_observed_at":"2026-08-06T06:00:03.766109Z","submitted_at":"2024-07-29T03:12:28Z","title":"mGTE: Generalized Long-Context Text Representation and Reranking Models for Multilingual Text Retrieval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.19669","snapshot_observed_at":"2026-08-07T11:26:07.174272Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:07.174272Z"},"links":{"cited_paper":"/paper/2407.19669","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:ecc56de3f7fbef183c51eb79cc0065020aeaa59a736a4412f0d06d9f926bcfef","observation_id":"873dcaaf-0b14-48e5-94f7-3d2ec62cc584","resolution":{"observed_at":"2026-08-07T11:26:07.174272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:07.266045Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:07.266045Z"},"links":{"citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:6fe202b76a756e447396be0e989a90e003d1435daaf85fefddc459e310b3b6a8","observation_id":"ffef1924-14c8-4f42-ac8b-2302eeb5382b","resolution":{"observed_at":"2026-08-07T11:26:07.266045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17631","last_updated":"2025-03-01T17:06:43Z","snapshot_observed_at":"2026-08-06T12:42:08.815092Z","submitted_at":"2023-10-26T17:48:58Z","title":"JudgeLM: Fine-tuned Large Language Models are Scalable Judges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17631","snapshot_observed_at":"2026-08-07T11:26:07.442164Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:07.442164Z"},"links":{"cited_paper":"/paper/2310.17631","citing_paper":"/paper/2506.02592"},"observation_digest":"sha256:aa7efcd568c0e2abcbc9214e61b2db162669aeb63c74f96698d26e4e122299b4","observation_id":"ec679e26-bc0e-4ae3-b46e-2637423e0d94","resolution":{"observed_at":"2026-08-07T11:26:07.442164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02592","last_updated":"2025-06-03T08:12:47Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T15:19:01.070855Z","submitted_at":"2025-06-03T08:12:47Z","title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":53,"verified_exact":2,"verified_fuzzy":0},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 3 inbound Pith citation observations for arXiv:2506.02592."}