{"as_of":"2026-08-08T02:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2e180504db8e82b43c7ecf60754677f75ce332b1e2b6374be0751e73f7a5fab6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":28,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:30:05.279668Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":6,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2408.15549","last_updated":"2026-04-17T16:47:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-28T05:53:46Z","title":"WildFeedback: Aligning LLMs With In-situ User Interactions And Feedback","version":4},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-23T22:37:43.230753Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2408.15549"},"observation_digest":"sha256:c305a96c4b1b9e9eed8f7c73ca4061d96742c2b133602385681e4a5c77ba2e33","observation_id":"96b355a5-e339-4e58-97a2-688587ece62d","resolution":{"observed_at":"2026-05-23T22:38:32.526940Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2410.20791","last_updated":"2026-04-06T19:29:25Z","snapshot_observed_at":"2026-08-03T01:41:19.546733Z","submitted_at":"2024-10-28T07:16:00Z","title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","version":3},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-05-23T19:07:21.016824Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2410.20791"},"observation_digest":"sha256:ce709087e8b16fe36d455d71792a00960846089782704bad129ef589c659f881","observation_id":"961e4675-97a3-4ad4-8d3b-a6b2d11d2e1a","resolution":{"observed_at":"2026-05-23T19:08:20.910382Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"reference_index":148,"source":"pdf_text","source_observed_at":"2026-05-23T17:33:13.394338Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2411.15594"},"observation_digest":"sha256:c562235358d42fbaf466ab07038a27b452177a9bd4985f699f982285a1e805d1","observation_id":"296e454e-1347-48f9-9d30-ecab492a6167","resolution":{"observed_at":"2026-05-23T17:35:44.000270Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2412.05579","last_updated":"2024-12-10T05:49:12Z","snapshot_observed_at":"2026-07-31T01:42:39.468673Z","submitted_at":"2024-12-07T08:07:24Z","title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","version":2},"reference_index":223,"source":"pdf_text","source_observed_at":"2026-05-11T23:08:34.312466Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2412.05579"},"observation_digest":"sha256:ce273221c1473238c7be9ab563ff41290be6e150a9fc137a35dbd3dde8809dcc","observation_id":"5bfeac36-f10b-4f39-9211-329636dd577a","resolution":{"observed_at":"2026-05-11T23:08:35.444375Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2502.16942","last_updated":"2025-06-02T07:51:11Z","snapshot_observed_at":"2026-07-06T20:41:28.562637Z","submitted_at":"2025-02-24T08:11:17Z","title":"NUTSHELL: A Dataset for Abstract Generation from Scientific Talks","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-23T02:50:10.307310Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2502.16942"},"observation_digest":"sha256:74deae854decbce2a3acd546c6c969860ec8788a4b969b78244bccc799a986d6","observation_id":"4bec3727-f646-495a-9b8f-975b8dd0f09d","resolution":{"observed_at":"2026-05-23T02:52:26.527289Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2504.14044","last_updated":"2025-04-18T19:24:17Z","snapshot_observed_at":"2026-08-03T12:16:52.455151Z","submitted_at":"2025-04-18T19:24:17Z","title":"Multi-Stage Retrieval for Operational Technology Cybersecurity Compliance Using Large Language Models: A Railway Casestudy","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-22T18:27:19.450554Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2504.14044"},"observation_digest":"sha256:bc9cff558e2a638163c11a0efffe2dca01a10faa46a345072c28cd89801f655d","observation_id":"47b452ea-8934-455f-80ad-a3fb6e8f80a8","resolution":{"observed_at":"2026-05-22T18:31:55.906972Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-07T05:30:05.279668Z","title":"Judging thejudges: Evaluatingalignmentand vulnerabilitiesin llms-as-judges","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07949","last_updated":"2025-06-09T17:14:41Z","snapshot_observed_at":"2026-08-07T05:19:16.640783Z","submitted_at":"2025-06-09T17:14:41Z","title":"Cost-Optimal Active AI Model Evaluation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:30:05.279668Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2506.07949"},"observation_digest":"sha256:bac37088f5660d240ab7dee169d2fa007a2bbcc34d2ed941586ac0e0bed1b227","observation_id":"47b3bed6-0941-4f76-8bd6-94c08b34f2d0","resolution":{"observed_at":"2026-08-07T05:30:05.279668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-07T05:01:08.310755Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09038","last_updated":"2025-06-10T17:57:30Z","snapshot_observed_at":"2026-08-07T04:54:28.526685Z","submitted_at":"2025-06-10T17:57:30Z","title":"AbstentionBench: Reasoning LLMs Fail on Unanswerable Questions","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-07T05:01:08.310755Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2506.09038"},"observation_digest":"sha256:7d77ce850cf0aa69ee25d66f860fb6be539a253d1623e8df8a4b34982426716a","observation_id":"730e3869-d1e8-479d-bd45-66c421fac6da","resolution":{"observed_at":"2026-08-07T05:01:08.310755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-07T04:07:07.823626Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.11825","last_updated":"2025-06-13T14:30:37Z","snapshot_observed_at":"2026-08-07T04:01:44.166136Z","submitted_at":"2025-06-13T14:30:37Z","title":"Revealing Political Bias in LLMs through Structured Multi-Agent Debate","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T04:07:07.823626Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2506.11825"},"observation_digest":"sha256:9f31d84d7e23b674b179aff7c0565c5023f90dfccf7ceccfeeacd11f9dc4b3da","observation_id":"bbcd6eb4-4ad1-4407-b1ff-d7cd2d376c3a","resolution":{"observed_at":"2026-08-07T04:07:07.823626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-07T00:31:36.353198Z","title":"Ramakrishna Vedantam, C","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.13639","last_updated":"2025-06-16T16:04:43Z","snapshot_observed_at":"2026-08-07T00:26:19.217552Z","submitted_at":"2025-06-16T16:04:43Z","title":"An Empirical Study of LLM-as-a-Judge: How Design Choices Impact Evaluation Reliability","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T00:31:36.353198Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2506.13639"},"observation_digest":"sha256:e817ef3fd9e254052c07b884d564ea57ffe4c669db814a1fbd891f15da1dad00","observation_id":"dbccbcef-a0e0-4a25-a3d0-e0ab3cfabc98","resolution":{"observed_at":"2026-08-07T00:31:36.353198Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-06T23:12:14.239923Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.19185","last_updated":"2025-06-23T23:02:57Z","snapshot_observed_at":"2026-08-06T23:04:45.394675Z","submitted_at":"2025-06-23T23:02:57Z","title":"Spiritual-LLM : Gita Inspired Mental Health Therapy In the Era of LLMs","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T23:12:14.239923Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2506.19185"},"observation_digest":"sha256:2ed544dfa73ee87ad9ac02de8fd54476e1f05039edf8854ab74c625f223ce7e4","observation_id":"6aeeb5cc-7968-4a71-ad09-0a8f7e651acb","resolution":{"observed_at":"2026-08-06T23:12:14.239923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-06T22:56:35.063138Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20274","last_updated":"2025-06-25T09:34:25Z","snapshot_observed_at":"2026-08-07T12:25:51.783364Z","submitted_at":"2025-06-25T09:34:25Z","title":"Enterprise Large Language Model Evaluation Benchmark","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T22:56:35.063138Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2506.20274"},"observation_digest":"sha256:d04e752bbb0b134fb0d96cc89c341627b3578dadfddef75ab0d15847d7eed3fa","observation_id":"2e9102f2-ea1c-4a32-b45b-7953a01a4fe7","resolution":{"observed_at":"2026-08-06T22:56:35.063138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-06T20:57:42.258308Z","title":"S., Choudhary, K., Ramayapally, V","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01413","last_updated":"2025-07-02T07:06:49Z","snapshot_observed_at":"2026-08-06T20:49:38.420684Z","submitted_at":"2025-07-02T07:06:49Z","title":"Evaluating LLM Agent Collusion in Double Auctions","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T20:57:42.258308Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2507.01413"},"observation_digest":"sha256:aab5d52065d9af33fc40adedef8fefe420ae040543bcd86761ab89ee37a05c19","observation_id":"65b8d7fa-94da-4a15-bcf1-17f2e4e41e66","resolution":{"observed_at":"2026-08-06T20:57:42.258308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T22:34:14.485583Z","title":"Judging the judges: Evaluating alignment and vulnerabili- ties in llms-as-judges,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06888","last_updated":"2025-08-09T08:35:40Z","snapshot_observed_at":"2026-08-07T20:00:53.865378Z","submitted_at":"2025-08-09T08:35:40Z","title":"Multi-Modal Requirements Data-based Acceptance Criteria Generation using LLMs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T22:34:14.485583Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2508.06888"},"observation_digest":"sha256:867afbfdb3f26ab7a708bf3462cb9dc3699656b9733e7aeefbbf7722723b2385","observation_id":"81e637f6-ca8b-4188-bcde-5563b1c381dd","resolution":{"observed_at":"2026-08-05T22:34:14.485583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T12:32:26.094061Z","title":"Judging the judges: Evaluating alignment and vulnerabilities in llms-as-judges,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.01494","last_updated":"2026-06-05T04:25:26Z","snapshot_observed_at":"2026-08-07T00:50:17.512250Z","submitted_at":"2025-09-01T14:13:34Z","title":"SWR-Bench: Assessing LLM Performance in Real-World Code Review Comment Generation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T12:32:26.094061Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2509.01494"},"observation_digest":"sha256:4e4c91cd555e388cc2b64a3d035db9c00ff2f3d8b26e75c2e91a40761791a923","observation_id":"be9185e6-f33f-4126-a79d-fa0c512e0962","resolution":{"observed_at":"2026-08-05T12:32:26.094061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-04T20:21:21.829738Z","title":"arXiv preprint arXiv:2406.12624","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08682","last_updated":"2025-09-10T15:22:00Z","snapshot_observed_at":"2026-08-07T12:17:08.646686Z","submitted_at":"2025-09-10T15:22:00Z","title":"Automatic Failure Attribution and Critical Step Prediction Method for Multi-Agent Systems Based on Causal Inference","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T20:21:21.829738Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2509.08682"},"observation_digest":"sha256:23d08b9de5fcea992446cdcaec54872267e7a41b1598511b616a4a0cf9f2e9ec","observation_id":"e6808b0f-c3d1-4656-8ec4-40373c23c594","resolution":{"observed_at":"2026-08-04T20:21:21.829738Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2510.00915","last_updated":"2026-05-22T12:16:11Z","snapshot_observed_at":"2026-08-02T22:13:21.681005Z","submitted_at":"2025-10-01T13:56:44Z","title":"Reinforcement Learning with Verifiable yet Noisy Rewards under Imperfect Verifiers","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-25T07:39:02.616294Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2510.00915"},"observation_digest":"sha256:1e4dc49e64debf5e1927efbfbfaae01de3b0a465d6c3afe9935800bddf034748","observation_id":"b6917440-946e-479c-b0d9-ecf3803b79ec","resolution":{"observed_at":"2026-05-25T07:40:28.682665Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2601.21464","last_updated":"2026-05-07T16:58:04Z","snapshot_observed_at":"2026-07-31T17:41:41.271363Z","submitted_at":"2026-01-29T09:41:14Z","title":"Conversation for Non-verifiable Learning: Self-Evolving LLMs through Meta-Evaluation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T10:04:44.379128Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2601.21464"},"observation_digest":"sha256:f794274f7f19feccd4139865362de954da12e7e67f017e56e1406189bee81fba","observation_id":"c79f583a-488a-4755-ab11-770e48249687","resolution":{"observed_at":"2026-05-16T10:07:43.327951Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-07-13T20:08:17.066898Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.22744","last_updated":"2026-05-29T00:35:02Z","snapshot_observed_at":"2026-08-03T18:04:48.244219Z","submitted_at":"2026-03-24T03:16:32Z","title":"LH-Bench: Skill-Grounded Evaluation of Long-Horizon Agents on Subjective Enterprise Tasks","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-13T20:08:17.066898Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2603.22744"},"observation_digest":"sha256:ee732db3f6309b1ce72c0406ae9849caa8db4532e598587f6d80455943c77ad8","observation_id":"155f65ea-abb3-4eec-bf17-4bce3d1a5f27","resolution":{"observed_at":"2026-07-13T20:08:17.066898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2603.23448","last_updated":"2026-04-07T08:07:13Z","snapshot_observed_at":"2026-07-06T22:50:23.931381Z","submitted_at":"2026-03-24T17:19:32Z","title":"Code Review Agent Benchmark","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T00:21:36.481731Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2603.23448"},"observation_digest":"sha256:beff09d7fe28bc423152b8f0b8ba5d8865266d2633dabafb9091503cffadc5f3","observation_id":"256cd814-938b-4ac8-9c5c-756b9e922b4a","resolution":{"observed_at":"2026-05-15T00:23:22.027618Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2604.16706","last_updated":"2026-04-17T21:15:35Z","snapshot_observed_at":"2026-07-06T23:03:57.199731Z","submitted_at":"2026-04-17T21:15:35Z","title":"Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T08:00:45.789649Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2604.16706"},"observation_digest":"sha256:b63a8a2d84312e55e860d9e344a0c12532053498c30754100b3c6c24c1a6b3de","observation_id":"98a09ffe-cc34-4299-8b87-cb3e37423673","resolution":{"observed_at":"2026-05-10T08:02:25.009796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-07-12T19:18:49.672005Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.16707","last_updated":"2026-06-21T14:58:53Z","snapshot_observed_at":"2026-08-03T04:47:14.528741Z","submitted_at":"2026-04-17T21:18:06Z","title":"Mixed response geometry and critical crossover in the Ising model","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-12T19:18:49.672005Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2604.16707"},"observation_digest":"sha256:cc29fab7b6c0bf172b4815b428de562b9b462aca7cc255f640a15f565d8f9b66","observation_id":"63347391-bfa9-4d3d-8b1c-439f5c3be01d","resolution":{"observed_at":"2026-07-12T19:18:49.672005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2604.24158","last_updated":"2026-04-27T08:13:57Z","snapshot_observed_at":"2026-07-06T23:10:16.426836Z","submitted_at":"2026-04-27T08:13:57Z","title":"Multi-Dimensional Evaluation of Sustainable City Trips with LLM-as-a-Judge and Human-in-the-Loop","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-08T03:32:38.940942Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2604.24158"},"observation_digest":"sha256:085d63e91040c2854a257a4309ba907d1a9ba9b56695e96464d94f0614059479","observation_id":"b5f7733b-4452-4167-be79-a2af72571ae4","resolution":{"observed_at":"2026-05-11T22:01:13.502497Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2605.03147","last_updated":"2026-05-04T20:40:05Z","snapshot_observed_at":"2026-07-06T23:16:05.372336Z","submitted_at":"2026-05-04T20:40:05Z","title":"Effective Performance Measurement: Challenges and Opportunities in KPI Extraction from Earnings Calls","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-08T18:04:50.972507Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2605.03147"},"observation_digest":"sha256:1cc06d9a04e1fe31a9bc5be582429ef462d4101c94a294d99d5b6d3407e6e4f6","observation_id":"60e1efee-b275-4e11-977a-a93f959dd882","resolution":{"observed_at":"2026-05-09T06:50:40.902897Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-02T13:54:55.570407Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:75cce503904e5d1f6f0985da54c2e5522e14960f2e43a7474397f95f20b5107c","observation_id":"071c21ff-3874-4b72-a93d-5e865503c0b0","resolution":{"observed_at":"2026-05-12T05:31:23.923870Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2605.28158","last_updated":"2026-05-27T08:41:30Z","snapshot_observed_at":"2026-07-06T23:37:45.223562Z","submitted_at":"2026-05-27T08:41:30Z","title":"OR-Space: A Full-Lifecycle Workspace Benchmark for Industrial Optimization Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T12:52:20.911788Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2605.28158"},"observation_digest":"sha256:dabe1be2b1533de34418c94078de42af0a326bd9c99bae5c8c4cf47f4795fd60","observation_id":"b5221e9f-b502-49db-967d-6016eb52bf13","resolution":{"observed_at":"2026-06-29T12:53:26.356204Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2606.02282","last_updated":"2026-06-01T14:05:35Z","snapshot_observed_at":"2026-07-06T23:42:44.661608Z","submitted_at":"2026-06-01T14:05:35Z","title":"POIROT: Interrogating Agents for Failure Detection in Multi-Agent Systems","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T14:44:21.487169Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2606.02282"},"observation_digest":"sha256:4133f6da557df8127ee5b4959990c42a2220ca4e2efc36ad6709140a122b8236","observation_id":"7233044c-0e63-43ce-a851-2f150da5f7d3","resolution":{"observed_at":"2026-07-01T23:06:19.960925Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","version":6},"cited_work":{"arxiv_id":"2406.12624","doi":"10.48550/arxiv.2406.12624","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.12624","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging the judges: Evaluating alignment and vulner- abilities in llms-as-judges","venue":"arXiv (Cornell University)","work_id":"d0e0145b-d0f4-43ff-8b70-fee617aaf5c5","year":2025},"citing_paper":{"arxiv_id":"2606.19714","last_updated":"2026-06-18T02:26:05Z","snapshot_observed_at":"2026-08-01T18:27:22.876701Z","submitted_at":"2026-06-18T02:26:05Z","title":"AURA: Adaptive Uncertainty-aware Refinement for LLM-as-a-Judge Auditing","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-26T15:48:26.303462Z"},"links":{"cited_paper":"/paper/2406.12624","citing_paper":"/paper/2606.19714"},"observation_digest":"sha256:9a02e836814215cc3d4df5de327fe279e41bf9794c840009dd0cfe6ae14b4df4","observation_id":"47b519b7-7810-4432-81d2-688802ec80e4","resolution":{"observed_at":"2026-07-04T05:39:40.098418Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2406.12624/citation-record","integrity":"/paper/2406.12624/integrity","json":"/paper/2406.12624/citation-record.json","paper":"/paper/2406.12624"},"outbound":[],"paper":{"arxiv_id":"2406.12624","last_updated":"2025-08-18T00:07:23Z","latest_version":6,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-01T14:58:17.402553Z","submitted_at":"2024-06-18T13:49:54Z","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 28 inbound Pith citation observations for arXiv:2406.12624."}