{"as_of":"2026-08-09T20:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bfca8e7542c7f0f0b635522de25a8bdbfd944b795bcebff371648055ea77902c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T16:14:57.387370Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2412.05579","last_updated":"2024-12-10T05:49:12Z","snapshot_observed_at":"2026-07-31T01:42:39.468673Z","submitted_at":"2024-12-07T08:07:24Z","title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","version":2},"reference_index":154,"source":"pdf_text","source_observed_at":"2026-05-11T23:08:34.312466Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2412.05579"},"observation_digest":"sha256:dfded340f3264b407877d0711c68ba4bfdb1fdf59cc9a53c91f466cb0fcabba4","observation_id":"a413f4a9-afcb-478d-8dcb-db1cbab5a1b4","resolution":{"observed_at":"2026-05-11T23:08:37.365470Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-08T16:14:57.387370Z","title":"Align- ing with human judgement: The role of pairwise preference in large language model evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.10441","last_updated":"2025-02-10T09:19:52Z","snapshot_observed_at":"2026-08-09T03:38:18.194039Z","submitted_at":"2025-02-10T09:19:52Z","title":"AI Alignment at Your Discretion","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-08T16:14:57.387370Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2502.10441"},"observation_digest":"sha256:cfccb4c6cd85478524113a7933529774c8216f8a32fa594d77e85e86f26752af","observation_id":"77254223-cb60-4048-bf6d-6f6806b002ac","resolution":{"observed_at":"2026-08-08T16:14:57.387370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-07T14:31:35.554333Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21537","last_updated":"2025-05-24T09:07:13Z","snapshot_observed_at":"2026-08-08T02:37:32.325481Z","submitted_at":"2025-05-24T09:07:13Z","title":"OpenReview Should be Protected and Leveraged as a Community Asset for Research in the Era of Large Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T14:31:35.554333Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2505.21537"},"observation_digest":"sha256:0e090505180a8ea0a52566139349f41f60384743e77db4d7f94e0fb119f11c88","observation_id":"045f3ca9-c561-4f3f-a762-ee3b14c03279","resolution":{"observed_at":"2026-08-07T14:31:35.554333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-07T01:06:45.759286Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13805","last_updated":"2025-06-13T17:12:47Z","snapshot_observed_at":"2026-08-09T15:07:28.749242Z","submitted_at":"2025-06-13T17:12:47Z","title":"Dr. GPT Will See You Now, but Should It? Exploring the Benefits and Harms of Large Language Models in Medical Diagnosis using Crowdsourced Clinical Cases","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T01:06:45.759286Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2506.13805"},"observation_digest":"sha256:d814166f69880464015126174d210eb2f53ee64bfb29ca369b6d47aa38bc0771","observation_id":"daa50822-7eaf-40f1-ba4a-31023c04a764","resolution":{"observed_at":"2026-08-07T01:06:45.759286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2506.14092","last_updated":"2026-07-23T04:06:28Z","snapshot_observed_at":"2026-08-08T11:55:38.025422Z","submitted_at":"2025-06-17T01:14:22Z","title":"Fragile Preferences: A Deep Dive Into Order Effects in Large Language Models","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-19T10:05:22.562737Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2506.14092"},"observation_digest":"sha256:b223ba2132e3cb9ce681e764c68c4e9e7a580a144d6f3feaf5021bb48db1fe78","observation_id":"02c6a2a9-05d9-4fdf-b7ce-ef5d54c8ffc8","resolution":{"observed_at":"2026-05-19T10:07:14.262841Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-07T00:25:01.132316Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14092","last_updated":"2026-07-23T04:06:28Z","snapshot_observed_at":"2026-08-08T11:55:38.025422Z","submitted_at":"2025-06-17T01:14:22Z","title":"Fragile Preferences: A Deep Dive Into Order Effects in Large Language Models","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:25:01.132316Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2506.14092"},"observation_digest":"sha256:55b15e4d0bfcd2556a8a07efccfa7c946f5a15dd07cbae49f0577d092c465cd7","observation_id":"45712385-d5db-491a-9802-e76798fa58b0","resolution":{"observed_at":"2026-08-07T00:25:01.132316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-06T15:24:21.497194Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16075","last_updated":"2025-07-21T21:23:21Z","snapshot_observed_at":"2026-08-07T12:20:24.237770Z","submitted_at":"2025-07-21T21:23:21Z","title":"Deep Researcher with Test-Time Diffusion","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T15:24:21.497194Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2507.16075"},"observation_digest":"sha256:3afadd74e247e8737b4bb7ca47c4b29fe189e6cdd558f6961f7d668bd12f277d","observation_id":"24071394-692b-4ee8-9aaf-90635da0d43c","resolution":{"observed_at":"2026-08-06T15:24:21.497194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T16:13:49.902821Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.18859","last_updated":"2025-08-26T09:38:29Z","snapshot_observed_at":"2026-08-06T02:26:11.795835Z","submitted_at":"2025-08-26T09:38:29Z","title":"Harnessing Meta-Learning for Controllable Full-Frame Video Stabilization","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-05T16:13:49.902821Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2508.18859"},"observation_digest":"sha256:47f8f3f440ed504917841585a2e4dee62a43b9725a70220fb695f9fa11acda17","observation_id":"9b4a0313-598a-45b9-a452-6ac597410690","resolution":{"observed_at":"2026-08-05T16:13:49.902821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2604.02655","last_updated":"2026-04-03T02:37:06Z","snapshot_observed_at":"2026-08-02T10:24:45.394666Z","submitted_at":"2026-04-03T02:37:06Z","title":"Semantic Data Processing with Holistic Data Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T19:07:49.756349Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2604.02655"},"observation_digest":"sha256:d405d4c02517ab0a04693ff4502e4cfd56ee078dbe1152b9818d704d7a170b5d","observation_id":"23d6e99e-59bb-4bdc-9aae-80085bd46aba","resolution":{"observed_at":"2026-05-13T19:08:09.534763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2604.16304","last_updated":"2026-01-25T10:36:59Z","snapshot_observed_at":"2026-07-06T23:03:39.351412Z","submitted_at":"2026-01-25T10:36:59Z","title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T11:26:38.634540Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2604.16304"},"observation_digest":"sha256:ec9d0c439b47bc97ffee9457ffad34a5ea8b51c6f43abe091c6f3ad3907f8744","observation_id":"333773b1-c3c1-434b-95d2-a90fbc04dafe","resolution":{"observed_at":"2026-05-16T11:27:48.133595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2604.23178","last_updated":"2026-06-24T13:27:28Z","snapshot_observed_at":"2026-08-08T12:19:05.889262Z","submitted_at":"2026-04-25T07:18:30Z","title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-08T08:14:18.535385Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2604.23178"},"observation_digest":"sha256:c6b985f99f5feb3135196b55934b38bcef4199a5790c0fa86c58812dfa64698b","observation_id":"a1076fd7-f506-448d-837a-fac402bdfa98","resolution":{"observed_at":"2026-05-11T20:41:14.214388Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2605.19141","last_updated":"2026-05-18T21:49:02Z","snapshot_observed_at":"2026-08-02T09:22:46.039410Z","submitted_at":"2026-05-18T21:49:02Z","title":"GRASP: Deterministic argument ranking in interaction graphs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-20T11:57:55.198779Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2605.19141"},"observation_digest":"sha256:04d05899b3372dd7d80849c385ed9f9e41a9e44c87d010c47f5451b8358cfbef","observation_id":"4ad14bc4-2452-4807-a925-15e9573003cc","resolution":{"observed_at":"2026-05-20T11:58:14.909207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2606.05384","last_updated":"2026-06-03T19:37:23Z","snapshot_observed_at":"2026-08-05T20:30:23.980379Z","submitted_at":"2026-06-03T19:37:23Z","title":"Stability vs. Manipulability: Evaluating Robustness Under Post-Decision Interaction in LLM Judges","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-06-28T05:58:59.870335Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2606.05384"},"observation_digest":"sha256:030c473dd0b6fc64e81bdd9ac687bc50272b8c0c3ba37388d7a1c1a7b67a35bc","observation_id":"a917ec0f-6780-4430-8a9c-2f8683a63e5d","resolution":{"observed_at":"2026-07-02T08:36:47.723043Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2606.24004","last_updated":"2026-06-29T07:01:12Z","snapshot_observed_at":"2026-08-02T11:15:29.427964Z","submitted_at":"2026-06-22T23:21:55Z","title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-26T07:49:36.816100Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2606.24004"},"observation_digest":"sha256:85fa834d59bfb079804a6fc9d70b81227424db012d2277a6bf9d7c560f2c5ecf","observation_id":"99177898-7b7c-45c6-9c21-ca4dc5678be1","resolution":{"observed_at":"2026-07-04T11:39:47.116638Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2606.24004","last_updated":"2026-06-29T07:01:12Z","snapshot_observed_at":"2026-08-02T11:15:29.427964Z","submitted_at":"2026-06-22T23:21:55Z","title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-30T10:17:33.176525Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2606.24004"},"observation_digest":"sha256:51f6e8384bf45ccc447d9240464574604c25ba53d0fdbc5b22943d3fbd33c94c","observation_id":"895e2b5d-14bb-41bc-aa26-1af41d739c06","resolution":{"observed_at":"2026-06-30T12:04:39.390666Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2606.27316","last_updated":"2026-06-25T17:29:58Z","snapshot_observed_at":"2026-07-31T08:21:54.248070Z","submitted_at":"2026-06-25T17:29:58Z","title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-26T03:56:28.271760Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2606.27316"},"observation_digest":"sha256:0db858e91c4dfa216a6b7f0c59f6b9660a3f6f27f6adc01c408f6f4266815006","observation_id":"62ed0cd6-59b0-4254-ba16-c3d9dd522f82","resolution":{"observed_at":"2026-06-26T03:58:57.117163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":"2403.16950","doi":"10.48550/arxiv.2403.16950","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Aligning with human judgement: The role of pairwise preference in large language model evaluators","venue":"arXiv (Cornell University)","work_id":"129a2f2d-3dba-4713-a819-b64f62aa6fd1","year":2025},"citing_paper":{"arxiv_id":"2606.27446","last_updated":"2026-06-25T18:17:04Z","snapshot_observed_at":"2026-08-08T06:45:59.827933Z","submitted_at":"2026-06-25T18:17:04Z","title":"Causal Connections: Leveraging Multilingual Fine-Tuning for Financial QA@FinCausal 2026","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-06-29T02:17:26.872854Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2606.27446"},"observation_digest":"sha256:7dd7e896d2bb5e396f1d78d4bb13272325bb298323091126ef9c1ceb11d683ef","observation_id":"4c09fb6f-9b4e-43ac-973a-10cf1f96a2c3","resolution":{"observed_at":"2026-06-29T02:23:00.942005Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T21:08:09.599674+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16950","snapshot_observed_at":"2026-07-31T12:20:07.163786Z","title":"Aligning with","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28282","last_updated":"2026-07-30T14:31:07Z","snapshot_observed_at":"2026-08-06T16:34:19.905250Z","submitted_at":"2026-07-30T14:31:07Z","title":"(Towards) Scalable Reliable Automated Evaluation with Large Language Models","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-07-31T12:20:07.163786Z"},"links":{"cited_paper":"/paper/2403.16950","citing_paper":"/paper/2607.28282"},"observation_digest":"sha256:fa61d3ffa52947c426a908fcbb3e70e06be82bf9a080274541bae781a5799db5","observation_id":"323dacc8-4392-4f0d-b5ad-4560fa5a6002","resolution":{"observed_at":"2026-07-31T12:20:07.163786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.16950/citation-record","integrity":"/paper/2403.16950/integrity","json":"/paper/2403.16950/citation-record.json","paper":"/paper/2403.16950"},"outbound":[],"paper":{"arxiv_id":"2403.16950","last_updated":"2025-01-17T03:43:53Z","latest_version":5,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T04:15:21.844410Z","submitted_at":"2024-03-25T17:11:28Z","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 18 inbound Pith citation observations for arXiv:2403.16950."}