{"as_of":"2026-08-05T09:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:94ad4dd5d95aa232beb42f529184f75f86e2d94f7800218db6b3d89c39502333","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T18:52:59.033645Z","state":"measured"},{"denominator":152,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":152,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":453,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T05:34:40.410309Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":8605,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2304.12244","last_updated":"2025-05-27T06:49:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-24T16:31:06Z","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-13T07:28:24.827546Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2304.12244"},"observation_digest":"sha256:642f78dedd9f04929ea32dc9b9dfdc9efcea1fed4dd17e4188a60b11c0868b89","observation_id":"f72bf196-212b-4c85-b3a9-b24f04886e8f","resolution":{"observed_at":"2026-05-13T07:28:25.016549Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2305.17926","last_updated":"2023-08-30T13:22:35Z","snapshot_observed_at":"2026-08-03T19:35:11.838629Z","submitted_at":"2023-05-29T07:41:03Z","title":"Large Language Models are not Fair Evaluators","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-17T12:10:42.248005Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2305.17926"},"observation_digest":"sha256:e4961e940d8d15135fde9c99ca2e63ffeda7b9d1c09f1935cb2bba84ae4c2a88","observation_id":"1367731e-1131-4f97-ba09-7c5979034c87","resolution":{"observed_at":"2026-05-17T12:10:42.345226Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-07-06T15:59:23.019044Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-24T07:42:09.112946Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2307.15043"},"observation_digest":"sha256:6a3fedfc49fa36ff8434859c0473ef6562f2551f3bce48192fc73ed331faca32","observation_id":"cebe60e5-147d-4845-8fa1-b8dd4a3d0d07","resolution":{"observed_at":"2026-05-24T07:44:08.450809Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2308.07201","last_updated":"2023-08-14T15:13:04Z","snapshot_observed_at":"2026-08-02T01:34:38.978920Z","submitted_at":"2023-08-14T15:13:04Z","title":"ChatEval: Towards Better LLM-based Evaluators through Multi-Agent Debate","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T13:03:18.765496Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2308.07201"},"observation_digest":"sha256:433bf6e14e44f8da249ea983ee6d1575ca0860c9d96157519ab399236185a655","observation_id":"f90ff1ce-b535-486d-8492-d0cfeb3e2748","resolution":{"observed_at":"2026-05-13T13:03:18.789065Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2308.14508","last_updated":"2024-06-19T04:00:32Z","snapshot_observed_at":"2026-08-02T11:20:36.216220Z","submitted_at":"2023-08-28T11:53:40Z","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","version":2},"reference_index":132,"source":"arxiv_source","source_observed_at":"2026-05-12T20:22:10.482509Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2308.14508"},"observation_digest":"sha256:ecf4ad1ec06f82c4b301df44e2bc352a86dc0d5ba86af54079c5798594ad0eb2","observation_id":"052242c2-f41a-4fe0-bc60-810a6b579599","resolution":{"observed_at":"2026-05-12T20:22:10.678903Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2309.05463","last_updated":"2023-09-11T14:01:45Z","snapshot_observed_at":"2026-08-02T22:47:03.212781Z","submitted_at":"2023-09-11T14:01:45Z","title":"Textbooks Are All You Need II: phi-1.5 technical report","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-14T19:18:01.244364Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2309.05463"},"observation_digest":"sha256:ce50fcf2c3cbc8d7f69d7ad6c8385c518a7631d590633101bc2c311f4cb39ec5","observation_id":"fc8959be-4c8c-47f3-a70c-3a5cda64fcdb","resolution":{"observed_at":"2026-05-14T19:18:01.272005Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2309.05653","last_updated":"2023-10-03T02:48:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-11T17:47:22Z","title":"MAmmoTH: Building Math Generalist Models through Hybrid Instruction Tuning","version":3},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-17T23:46:39.330438Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2309.05653"},"observation_digest":"sha256:5b568185afda2de0d4582ee23464e6328b3eca74bc5a7a1eb277ffec44c95094","observation_id":"f0c63962-c1af-4c2e-bb0e-710a0372b3d7","resolution":{"observed_at":"2026-05-17T23:46:39.656638Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2309.10253","last_updated":"2024-06-27T16:01:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-19T02:19:48Z","title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","version":4},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-15T06:25:20.966510Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2309.10253"},"observation_digest":"sha256:abe78308051d5a2f97093f7cf1426a715f917572a3c03c340eefa2c3850f0265","observation_id":"3130017a-da2d-49d2-bf03-33ddd3fade96","resolution":{"observed_at":"2026-05-15T06:25:21.165188Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2309.11381","last_updated":"2023-09-20T15:03:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-20T15:03:30Z","title":"Studying Lobby Influence in the European Parliament","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-24T06:40:16.620051Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2309.11381"},"observation_digest":"sha256:50d6b98c7ecdbd55123d7824550e22d06e631de674b2a3dc528b6305bb57f63a","observation_id":"5e2d0391-945c-4c1d-a43e-124be978a2cb","resolution":{"observed_at":"2026-05-24T06:44:02.430659Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2310.00754","last_updated":"2024-03-16T19:28:08Z","snapshot_observed_at":"2026-08-02T06:11:56.496482Z","submitted_at":"2023-10-01T18:10:53Z","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T22:46:52.791128Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2310.00754"},"observation_digest":"sha256:89c4dc2ebe08edf27d49ffa2129eddaa022fde867ad22c9c63cc465ba82a6b5f","observation_id":"d6224d4d-b0d8-40d9-9432-72ac74903836","resolution":{"observed_at":"2026-05-17T22:46:52.841020Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2310.03693","last_updated":"2023-10-05T17:12:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-05T17:12:17Z","title":"Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T08:58:35.714394Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2310.03693"},"observation_digest":"sha256:bb84e4c2b40dbf965e708caead9c747ffbaddfbc69fe1bbd79b6bcf0a3de2d5d","observation_id":"dd725577-172a-4ad1-8c24-ae36a7cc700c","resolution":{"observed_at":"2026-05-12T08:58:35.770758Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2310.08560","last_updated":"2024-02-12T18:59:46Z","snapshot_observed_at":"2026-08-03T06:31:15.541423Z","submitted_at":"2023-10-12T17:51:32Z","title":"MemGPT: Towards LLMs as Operating Systems","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T12:27:29.041352Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2310.08560"},"observation_digest":"sha256:f77910c809a932cec016f2cd204634dbfb69bcd4aaa99fc73fc728caea270b4e","observation_id":"fc576694-5a01-4e56-becc-416091abdec1","resolution":{"observed_at":"2026-05-10T18:52:59.240297Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2310.16944","last_updated":"2023-10-25T19:25:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-25T19:25:16Z","title":"Zephyr: Direct Distillation of LM Alignment","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-16T10:13:57.361932Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2310.16944"},"observation_digest":"sha256:9ed6bf3a9b4c30b703ddf0b288d9454b6ff6df47c87056971b3e5b23edd32087","observation_id":"cdf31227-10da-489d-9165-4cfaf1e39177","resolution":{"observed_at":"2026-05-16T10:13:57.426351Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2311.16867","last_updated":"2023-11-29T19:45:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-28T15:12:47Z","title":"The Falcon Series of Open Language Models","version":2},"reference_index":203,"source":"arxiv_source","source_observed_at":"2026-05-16T09:46:09.701440Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2311.16867"},"observation_digest":"sha256:c9e1030b947b6c2c00e78595b0db7ce67992064f82abf17c1a80466f11a87fd8","observation_id":"b12c085f-27ff-4366-8209-095acbc4c3e2","resolution":{"observed_at":"2026-05-16T09:46:09.880853Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2312.08935","last_updated":"2024-02-19T14:07:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-14T13:41:54Z","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","version":3},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-05-14T22:34:15.638114Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2312.08935"},"observation_digest":"sha256:8e924af80ae085af84dad09938eb34b1b4974470f4273f55716d882479306b7f","observation_id":"088c9e89-3ce7-410e-8745-03961d61da6b","resolution":{"observed_at":"2026-05-14T22:34:15.895133Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2312.13771","last_updated":"2026-07-05T07:51:04Z","snapshot_observed_at":"2026-08-04T19:11:48.797552Z","submitted_at":"2023-12-21T11:52:45Z","title":"AppAgent: Multimodal Agents as Smartphone Users","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-17T10:16:43.364787Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2312.13771"},"observation_digest":"sha256:fbf4726e5216a3370dc8d92ad24d653c183fe52ff3ca4b64cccf9ca01ec04766","observation_id":"196a7187-8552-4dc3-bc31-ac0d35994c4b","resolution":{"observed_at":"2026-05-17T10:16:43.891325Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"reference_index":185,"source":"pdf_text","source_observed_at":"2026-05-13T22:46:09.693156Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2312.14238"},"observation_digest":"sha256:35ab5ce615d1cd929a9f9e8e3901ddbb2f22b93b24391ec34b2fd94241fb5571","observation_id":"0c706940-5785-426b-8f50-4b8b46b1b738","resolution":{"observed_at":"2026-05-13T22:46:10.270145Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2312.16886","last_updated":"2023-12-30T04:59:21Z","snapshot_observed_at":"2026-07-29T22:12:01.108911Z","submitted_at":"2023-12-28T08:21:24Z","title":"MobileVLM : A Fast, Strong and Open Vision Language Assistant for Mobile Devices","version":2},"reference_index":133,"source":"pdf_text","source_observed_at":"2026-05-16T16:35:37.937462Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2312.16886"},"observation_digest":"sha256:d7cfa635d558d26719178ed0444334ece09db2f2c106635a5a208403bd057c40","observation_id":"27a7750c-c8d6-48be-99e3-9f2371b8a217","resolution":{"observed_at":"2026-05-16T16:35:38.191524Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-24T04:09:15.921778Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2401.04088"},"observation_digest":"sha256:fd232fce463198de884d17774cbfb509fb5afd9ef062af73ddb89ded562be525","observation_id":"f2032401-9541-4524-aaa3-25603ee12505","resolution":{"observed_at":"2026-05-24T04:13:53.885898Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2401.15077","last_updated":"2025-03-04T13:58:39Z","snapshot_observed_at":"2026-08-03T09:40:31.365295Z","submitted_at":"2024-01-26T18:59:01Z","title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","version":3},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-15T00:15:49.303458Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2401.15077"},"observation_digest":"sha256:63ea1fe07fad64eb7a3da7f448f8cb3abfd770ae8ab03823bed41b0047f2d474","observation_id":"39d1a7cb-912f-439d-9acd-bd67ad725d2d","resolution":{"observed_at":"2026-05-15T00:15:49.438354Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2402.01306","last_updated":"2024-11-19T18:12:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-02T10:53:36Z","title":"KTO: Model Alignment as Prospect Theoretic Optimization","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T12:17:53.478052Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2402.01306"},"observation_digest":"sha256:e823e6c31d39fd13f0a4dc4da7023f2759230d2d49caaa0701cdf3b4d0de2a1d","observation_id":"0c8235e4-cde4-4e2e-9794-3495b8e7a912","resolution":{"observed_at":"2026-05-12T12:17:53.549452Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2402.08268","last_updated":"2025-02-03T21:47:31Z","snapshot_observed_at":"2026-08-01T14:30:18.054312Z","submitted_at":"2024-02-13T07:47:36Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-16T06:36:57.165551Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2402.08268"},"observation_digest":"sha256:e35cf8fd3c0a407dc09e397757c013307b731334925508b31e97abdaa90b60de","observation_id":"feda9130-605b-45b9-9ae9-89e7d424b270","resolution":{"observed_at":"2026-05-16T06:36:57.248655Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2402.13228","last_updated":"2024-07-03T13:46:33Z","snapshot_observed_at":"2026-07-31T05:05:41.080329Z","submitted_at":"2024-02-20T18:42:34Z","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-17T23:04:44.287660Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2402.13228"},"observation_digest":"sha256:48d78a96d0491802949e4fd3e37141104a3257105479fea4f56b56cab66da957","observation_id":"d468840c-4596-427c-ad01-6ce08cc3c3bc","resolution":{"observed_at":"2026-05-17T23:04:44.457086Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2403.07691","last_updated":"2024-03-14T07:47:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T14:34:08Z","title":"ORPO: Monolithic Preference Optimization without Reference Model","version":2},"reference_index":137,"source":"arxiv_source","source_observed_at":"2026-05-16T09:34:04.394588Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2403.07691"},"observation_digest":"sha256:5403e6f282bf9f26793ba9273fafd035d33fcffe254311ca6017c56b3baf0c23","observation_id":"0326ac5c-d92f-4a0b-aafa-749f459f01c5","resolution":{"observed_at":"2026-05-16T09:34:04.516226Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2403.12031","last_updated":"2024-03-28T17:56:28Z","snapshot_observed_at":"2026-07-06T17:46:29.153377Z","submitted_at":"2024-03-18T17:59:04Z","title":"RouterBench: A Benchmark for Multi-LLM Routing System","version":2},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-05-16T10:47:31.006944Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2403.12031"},"observation_digest":"sha256:d284503ff2e673c0cb8805a380b9222dd6ff90958cc1d48a0bb94051fbf3abc1","observation_id":"528933f3-c0ec-4b08-849a-5ea1f4c09bc4","resolution":{"observed_at":"2026-05-16T10:47:31.106113Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2404.01318","last_updated":"2024-10-31T22:26:40Z","snapshot_observed_at":"2026-08-02T14:59:12.115203Z","submitted_at":"2024-03-28T02:44:02Z","title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","version":5},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-15T06:08:05.386345Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2404.01318"},"observation_digest":"sha256:83eaafc787fe0ce075584aa1639fdc16c38679a59a426c3131ccd20adfe6bc66","observation_id":"cd42278d-1978-49f0-81df-f7c053c8745c","resolution":{"observed_at":"2026-05-15T06:08:05.546872Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T20:19:27.255515Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2404.14219"},"observation_digest":"sha256:e6a0344286c554e0367738371a81bb51fc0c9a26b36cecc14d458818d95cb019","observation_id":"95b35164-7faa-4f42-a8a7-fac2463f6dc5","resolution":{"observed_at":"2026-05-10T20:19:27.284354Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2405.14782","last_updated":"2026-05-31T00:04:33Z","snapshot_observed_at":"2026-08-02T05:44:43.939336Z","submitted_at":"2024-05-23T16:50:49Z","title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-16T18:44:49.519995Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2405.14782"},"observation_digest":"sha256:5b396cec5035bef542c2994d1be73e3993caf542a895b0322e8c73da7769976a","observation_id":"71f63490-e1d6-4585-b818-c0693bd39714","resolution":{"observed_at":"2026-05-16T18:44:49.614200Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"reference_index":273,"source":"arxiv_source","source_observed_at":"2026-05-18T06:38:36.517935Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2408.00724"},"observation_digest":"sha256:9404594b442794095c53e6049aeca9546295c457f79d1d05ee2d2873dc64a134","observation_id":"3cb9f199-ad53-4cca-ae65-d2774ebd1ba3","resolution":{"observed_at":"2026-05-18T06:38:37.117555Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2408.15339","last_updated":"2026-05-07T20:32:48Z","snapshot_observed_at":"2026-08-02T15:53:33.114051Z","submitted_at":"2024-08-27T18:04:07Z","title":"UNA: A Unified Supervised Framework for Efficient LLM Alignment Across Feedback Types","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-23T21:22:36.970101Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2408.15339"},"observation_digest":"sha256:18652e27acb29cae40cd2c9dc8fc7495604b4ba6901ea1203a0fbd4d1bb1dc81","observation_id":"737e8445-6ce5-4896-920a-334a477100e6","resolution":{"observed_at":"2026-05-23T21:23:27.447863Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2408.15549","last_updated":"2026-04-17T16:47:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-28T05:53:46Z","title":"WildFeedback: Aligning LLMs With In-situ User Interactions And Feedback","version":4},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-23T22:37:43.230753Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2408.15549"},"observation_digest":"sha256:cc167037abc4053d6f3d269c3295cc38f3cac384cc42c9cf6169e04636765838","observation_id":"8e5d07f4-d36f-483e-9df5-acce550ee1ba","resolution":{"observed_at":"2026-05-23T22:38:32.597717Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2412.02612","last_updated":"2024-12-03T17:41:24Z","snapshot_observed_at":"2026-08-02T21:27:26.884251Z","submitted_at":"2024-12-03T17:41:24Z","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T03:53:47.396742Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2412.02612"},"observation_digest":"sha256:ba42c943cd054ca007a665655e1cfb54994ee4ee16b386382d7c8b5ac4d12744","observation_id":"13dd085b-36e2-49f1-b5c6-9e8b5f9fdc4b","resolution":{"observed_at":"2026-05-16T03:53:47.574522Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2503.13657","last_updated":"2025-10-26T23:25:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-17T19:04:38Z","title":"Why Do Multi-Agent LLM Systems Fail?","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T05:42:57.561468Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2503.13657"},"observation_digest":"sha256:55b4e42278bee19006fdc058ce1ee17242ae00dc16722439ef9505c7ef740a7a","observation_id":"9315eda1-2d6a-4f23-9543-8901b8611b92","resolution":{"observed_at":"2026-05-12T05:42:58.153412Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2504.16155","last_updated":"2026-05-07T09:59:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:52:04Z","title":"PRIMETIME : Limits of LLMs in Temporal Primitives","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-22T18:36:48.376877Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2504.16155"},"observation_digest":"sha256:70b549fcb9a80b25cd848e186718d2948bf5cd4f57b45241d10b4e31a4fe0a92","observation_id":"43e780d3-cafa-43ff-98be-628f5a5392bd","resolution":{"observed_at":"2026-05-22T18:36:58.774117Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2504.20605","last_updated":"2026-05-02T05:59:29Z","snapshot_observed_at":"2026-07-06T21:16:23.106510Z","submitted_at":"2025-04-29T10:15:28Z","title":"TF1-EN-3M: Three Million Synthetic Moral Fables for Training Small, Open Language Models","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-22T19:01:42.307514Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2504.20605"},"observation_digest":"sha256:88a6eff90aa6441c5813e8adcc5961cc338a4cd4cc9208e1c78bb77a7c24919f","observation_id":"240eb6cb-23e8-4411-b9be-1ccc02f26069","resolution":{"observed_at":"2026-05-22T19:01:57.820004Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-02T16:13:31.498470Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"reference_index":178,"source":"pdf_text","source_observed_at":"2026-05-11T05:26:04.960844Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2505.07062"},"observation_digest":"sha256:3782e4df00fc8ce3217141f3af52a814e3b410cb99da13b4dbfeaca191abd2c9","observation_id":"82c6491b-073d-48cd-9408-004033b0667d","resolution":{"observed_at":"2026-05-11T05:26:05.911342Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2505.17682","last_updated":"2026-04-13T03:44:31Z","snapshot_observed_at":"2026-08-02T09:49:15.593615Z","submitted_at":"2025-05-23T09:53:43Z","title":"Tuning Language Models for Robust Prediction of Diverse User Behaviors","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-19T13:51:21.147668Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2505.17682"},"observation_digest":"sha256:e8b5ee2f52bdf6c022d8d4892e385aa8c49b212a82e75d4d61c00a746288c73b","observation_id":"db66ab1b-95a1-4555-9462-5ab75a143a43","resolution":{"observed_at":"2026-05-19T13:52:19.790722Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2505.19237","last_updated":"2026-04-19T18:56:45Z","snapshot_observed_at":"2026-07-06T21:30:09.658226Z","submitted_at":"2025-05-25T17:26:28Z","title":"Sensorimotor Self-Recognition in Multimodal Large Language Model-Driven Robots","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-19T13:29:34.151546Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2505.19237"},"observation_digest":"sha256:814bcc0680431b7739f62182d80b16b705e78bd3f50324c0bbdd8939ff95f24c","observation_id":"c52f11c7-9882-45bf-87f5-182aa1952a75","resolution":{"observed_at":"2026-05-19T13:32:19.305119Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2505.20340","last_updated":"2026-05-02T12:41:59Z","snapshot_observed_at":"2026-07-06T21:30:52.246162Z","submitted_at":"2025-05-24T14:17:50Z","title":"Latent Trajectory Dynamics in Large Language Models: A Manifold Evolution Framework with Empirical Validation","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-19T12:55:31.950743Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2505.20340"},"observation_digest":"sha256:9109b2db5d920a054eed8b46437b4e3cd175a0ca266944d14a59c73a891fd960","observation_id":"feb105f9-173b-4e0e-9f53-991e3a26dde0","resolution":{"observed_at":"2026-05-19T12:57:17.804573Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2505.21472","last_updated":"2025-08-11T19:24:13Z","snapshot_observed_at":"2026-07-06T21:31:35.572708Z","submitted_at":"2025-05-27T17:45:21Z","title":"Mitigating Hallucination in Large Vision-Language Models via Adaptive Attention Calibration","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T12:48:44.324236Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2505.21472"},"observation_digest":"sha256:58d0f4a4a0e867b1ddbb9e95d6e788c14e916accf7c690bd865e8a1d5ace3b44","observation_id":"1041e806-0c87-4455-a0d6-e4349fdc1a2f","resolution":{"observed_at":"2026-05-19T12:52:18.035155Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2505.21972","last_updated":"2026-04-05T18:32:59Z","snapshot_observed_at":"2026-07-06T21:31:54.566899Z","submitted_at":"2025-05-28T04:50:41Z","title":"LLMs Judging LLMs: A Simplex Perspective","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-19T12:40:37.816449Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2505.21972"},"observation_digest":"sha256:1a9f59665089c45469ef3a922d07574ae37a5681c55733d873b6a2b4f07bcb6b","observation_id":"464ecab5-6583-4867-a904-ba5f6ed32fac","resolution":{"observed_at":"2026-05-19T12:42:18.514994Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2506.11763","last_updated":"2025-06-13T13:17:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-13T13:17:32Z","title":"DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T08:07:39.384613Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2506.11763"},"observation_digest":"sha256:4b18f912b024cc98817e65405ee83835834af5b3c17dfc51deb0a9744b4dca83","observation_id":"9c4b1a50-88ea-4461-944b-2f7ed61a1caa","resolution":{"observed_at":"2026-05-16T08:07:39.536246Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2506.22832","last_updated":"2026-04-09T11:44:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-28T09:53:17Z","title":"Listener-Rewarded Thinking in VLMs for Image Preferences","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-19T07:38:52.273903Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2506.22832"},"observation_digest":"sha256:11e9e613118248b1baa7a72aa4f782467aabf0ce3aadec031ce4904d862e0afa","observation_id":"0f794dd2-4430-4638-b7e0-a2e529df5015","resolution":{"observed_at":"2026-05-19T07:42:09.208655Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-19T05:48:02.828938Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2507.06261"},"observation_digest":"sha256:b0cca1921611fb757dfa322dad80a7dc8732e595bbd403821855758892319ebd","observation_id":"314c7ca6-8ba2-4e65-85ea-e43f7ebea27e","resolution":{"observed_at":"2026-05-19T05:52:07.707421Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2508.11011","last_updated":"2026-05-26T21:38:21Z","snapshot_observed_at":"2026-08-01T15:13:30.605711Z","submitted_at":"2025-08-14T18:23:09Z","title":"Are Large Pre-trained Vision Language Models Effective Construction Safety Inspectors?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-18T22:32:11.401201Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2508.11011"},"observation_digest":"sha256:38ad32cc9a82e8a08f17606c299a1140f7c4b50f4a68520fc2ae190b7c8d1945","observation_id":"4b4852fc-c0e4-496a-bfd9-8c4fe01178ae","resolution":{"observed_at":"2026-05-18T22:32:52.366733Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2508.15202","last_updated":"2026-05-04T16:17:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-21T03:31:11Z","title":"Fin-PRM: A Domain-Specialized Process Reward Model for Financial Reasoning in Large Language Models","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-18T22:41:26.047957Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2508.15202"},"observation_digest":"sha256:5f1d9e5b6011edf815d98b560ae84b3b5e1be96282da7f696faf1234e7a65049","observation_id":"929cbaa5-cc58-4f21-b1da-598553038c49","resolution":{"observed_at":"2026-05-18T22:41:53.055375Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2508.15919","last_updated":"2026-04-23T22:48:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-21T18:40:20Z","title":"HFX: Joint Design of Algorithms and Systems for Multi-SLO Serving and Fast Scaling","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-18T21:39:02.560962Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2508.15919"},"observation_digest":"sha256:e1071d0985c4b2ea522f9b6e5a6c645ead43cdbfa34d5b5650ac3fbc13f64d1d","observation_id":"ab01ebc0-fa6f-48e0-b1f0-92babacb6e70","resolution":{"observed_at":"2026-05-18T21:41:51.709441Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T05:34:40.410309Z","title":"Zheng, W.-L","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.05226","last_updated":"2025-09-05T16:40:13Z","snapshot_observed_at":"2026-08-05T05:34:39.068793Z","submitted_at":"2025-09-05T16:40:13Z","title":"Less is More Tokens: Efficient Math Reasoning via Difficulty-Aware Chain-of-Thought Distillation","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-05T05:34:40.410309Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.05226"},"observation_digest":"sha256:df7d82edaa5c9815fbb9914a9d43b98a2efed265dcae76dab0f552190c0e0705","observation_id":"9bf28a27-842f-47d5-be4f-26a3a1b8ddb3","resolution":{"observed_at":"2026-08-05T05:34:40.410309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T23:49:34.834441Z","title":"P.; Zhang, H.; Gonzalez, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.06350","last_updated":"2026-01-27T23:14:36Z","snapshot_observed_at":"2026-08-04T23:49:28.733609Z","submitted_at":"2025-09-08T05:45:37Z","title":"Mask-GCG: Are All Tokens in Adversarial Suffixes Necessary for Jailbreak Attacks?","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-04T23:49:34.834441Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.06350"},"observation_digest":"sha256:c798f010c6d3b3d8ac0cfb62726c4bb6449ad575e9d9a177748cb241c818846d","observation_id":"e8de947b-8b25-4cb3-871d-ecce4390806c","resolution":{"observed_at":"2026-08-04T23:49:34.834441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T23:09:41.423390Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.06807","last_updated":"2025-09-08T15:39:17Z","snapshot_observed_at":"2026-08-04T23:09:39.534978Z","submitted_at":"2025-09-08T15:39:17Z","title":"MoGU V2: Toward a Higher Pareto Frontier Between Model Usability and Security","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T23:09:41.423390Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.06807"},"observation_digest":"sha256:197bf9985bbb9be830b082055520ace6dca91bb94681054b846ff7820ea67b53","observation_id":"edc29ab5-a9a5-4367-a3c0-04ee523c00e2","resolution":{"observed_at":"2026-08-04T23:09:41.423390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2509.07177","last_updated":"2026-04-14T15:07:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-08T19:48:52Z","title":"Towards EnergyGPT: A Large Language Model Specialized for the Energy Sector","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-18T17:39:17.456350Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.07177"},"observation_digest":"sha256:45bd2ab730e6b9cb12948452ac8f3205fc118d5725889c42ca6434fe2c0e4038","observation_id":"005b982f-aff8-4b26-92a0-123ac3976a19","resolution":{"observed_at":"2026-05-18T17:42:46.082416Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T20:04:29.442012Z","title":"Judging LLM-as-a- judge with MT-bench and chatbot arena,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08910","last_updated":"2025-09-10T18:14:52Z","snapshot_observed_at":"2026-08-04T20:04:28.553952Z","submitted_at":"2025-09-10T18:14:52Z","title":"PromptGuard: An Orchestrated Prompting Framework for Principled Synthetic Text Generation for Vulnerable Populations using LLMs with Enhanced Safety, Fairness, and Controllability","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-04T20:04:29.442012Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.08910"},"observation_digest":"sha256:7dd90457da8dbb7dbaa47c310c2639f21555e69900dbd8cbbb836d505e9efb57","observation_id":"963681bd-0677-4864-9bc4-8c316c747d66","resolution":{"observed_at":"2026-08-04T20:04:29.442012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T23:35:05.545669Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09714","last_updated":"2025-09-08T11:00:18Z","snapshot_observed_at":"2026-08-04T23:35:00.567866Z","submitted_at":"2025-09-08T11:00:18Z","title":"How Small Transformation Expose the Weakness of Semantic Similarity Measures","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-04T23:35:05.545669Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.09714"},"observation_digest":"sha256:cf71d70c2f6e9d6588e1d4bc9544e25ce6f938eb82e2f9829e12c0844f460589","observation_id":"1d55773f-5807-4f6a-b8e0-4bb875b9df02","resolution":{"observed_at":"2026-08-04T23:35:05.545669Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T18:37:55.129913Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09852","last_updated":"2025-09-11T21:01:54Z","snapshot_observed_at":"2026-08-04T18:37:47.101607Z","submitted_at":"2025-09-11T21:01:54Z","title":"Topic-Guided Reinforcement Learning with LLMs for Enhancing Multi-Document Summarization","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-04T18:37:55.129913Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.09852"},"observation_digest":"sha256:c5795ca2a3ca90695b29fad9d6e5f6d3f500bc9324d75ca9a6afe7c39df9eef5","observation_id":"6b02f4cf-8c24-475a-8010-0af7e54faa1b","resolution":{"observed_at":"2026-08-04T18:37:55.129913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T17:15:27.898978Z","title":"Faithfulness","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.11026","last_updated":"2025-09-14T01:33:14Z","snapshot_observed_at":"2026-08-04T17:15:27.499140Z","submitted_at":"2025-09-14T01:33:14Z","title":"Rethinking Human Preference Evaluation of LLM Rationales","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T17:15:27.898978Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.11026"},"observation_digest":"sha256:ac33bf3a9ac6cbf7d1f0ed00e0e38d6de5f7ba22b29d3ea1d5304c25dd0fea9a","observation_id":"9a58348b-4a58-4cc3-ad89-25bfcc4479f0","resolution":{"observed_at":"2026-08-04T17:15:27.898978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T17:10:42.817913Z","title":"Available: https://arxiv.org/abs/2306.05685","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.11068","last_updated":"2025-09-14T03:30:06Z","snapshot_observed_at":"2026-08-05T01:26:24.301315Z","submitted_at":"2025-09-14T03:30:06Z","title":"Tractable Asymmetric Verification for Large Language Models via Deterministic Replicability","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T17:10:42.817913Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.11068"},"observation_digest":"sha256:8c1335357062043ecd44e3c1738e71b15d281d1ef1c8e8a9181470a523052fb4","observation_id":"228824bd-89a2-410c-b3e9-c9a035ef7786","resolution":{"observed_at":"2026-08-04T17:10:42.817913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2509.11206","last_updated":"2026-04-20T05:43:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-14T10:24:13Z","title":"Evalet: Evaluating Large Language Models through Functional Fragmentation","version":4},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-05-18T16:57:25.259866Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.11206"},"observation_digest":"sha256:9fab30033793828deaa6f01b1fda319ebc96035e99de23ecfb9ffcf021edc81c","observation_id":"012c0adc-9f11-47b1-93ed-1f892e9ce70a","resolution":{"observed_at":"2026-05-18T17:01:39.969925Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T17:46:56.345107Z","title":"P Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.12263","last_updated":"2026-05-31T19:40:08Z","snapshot_observed_at":"2026-08-04T17:46:52.771450Z","submitted_at":"2025-09-12T20:07:12Z","title":"InPhyRe Discovers: Large Multimodal Models Struggle in Inductive Physical Reasoning","version":3},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-04T17:46:56.345107Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.12263"},"observation_digest":"sha256:4aab5176ded87fb178fde8d1d01625246ce64130e82d9f700f34204f2b3b3ae8","observation_id":"009d1fb1-6d35-46d2-92bf-eef58b50ac02","resolution":{"observed_at":"2026-08-04T17:46:56.345107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2509.26383","last_updated":"2026-05-22T06:49:31Z","snapshot_observed_at":"2026-08-02T17:45:26.442638Z","submitted_at":"2025-09-30T15:14:24Z","title":"Efficient and Transferable Agentic Knowledge Graph RAG via Reinforcement Learning","version":5},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-25T07:44:05.582003Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2509.26383"},"observation_digest":"sha256:d800a089d035d3166d2d717360e73598de281a3c3a8966ed7f69be58eff987fe","observation_id":"fb320e90-b39c-420a-97c4-507f3a07e8b1","resolution":{"observed_at":"2026-05-25T07:45:28.889091Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T13:26:50.860897Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.00501","last_updated":"2026-06-22T13:28:57Z","snapshot_observed_at":"2026-08-05T05:24:07.258585Z","submitted_at":"2025-10-01T04:33:53Z","title":"CodeChemist: Functional Knowledge Transfer for Low-Resource Code Generation via Test-Time Scaling","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T13:26:50.860897Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2510.00501"},"observation_digest":"sha256:261fed1fae2f30cc70910ff3705fb83a85abf86dbc3795ae976dfef90c7122b0","observation_id":"5c8ed9b6-98ce-49dd-a0d8-ef30cdfba3ce","resolution":{"observed_at":"2026-08-04T13:26:50.860897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2510.04056","last_updated":"2026-05-12T17:38:57Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T06:34:30Z","title":"QuiLL: An LLM-Based Vulnerability Assessment Framework for the Wild","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-18T10:56:28.973065Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2510.04056"},"observation_digest":"sha256:0cee43ecbc61072b3d411c8dc4a4e408676ef3f06bd30b0daeff964f3b2d8b90","observation_id":"e6c14c3a-3b70-40cb-95c9-78c013e4aab9","resolution":{"observed_at":"2026-05-18T11:01:17.190998Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2510.04265","last_updated":"2026-05-12T01:55:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-05T16:14:03Z","title":"Don't Pass@k: A Bayesian Framework for Large Language Model Evaluation","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-18T10:04:39.223895Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2510.04265"},"observation_digest":"sha256:aceac103e80ec6c9f77f1bf8673fb7ee05644936bba9305088518cec9e7e8c19","observation_id":"b7456044-d828-481a-ab2c-7a238b0c30d3","resolution":{"observed_at":"2026-05-18T10:06:13.702902Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T11:09:41.074310Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.08622","last_updated":"2026-06-09T10:52:38Z","snapshot_observed_at":"2026-08-04T11:09:33.510027Z","submitted_at":"2025-10-08T09:12:57Z","title":"Automated Alignment between Elicitation Interviews and Requirements","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-04T11:09:41.074310Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2510.08622"},"observation_digest":"sha256:40d34429b498c7c1f1145477c260e86743e227367fa59aee6f2b674ca78aa569","observation_id":"fe30e200-780e-4307-b31f-fa45eebefbc3","resolution":{"observed_at":"2026-08-04T11:09:41.074310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2510.11194","last_updated":"2026-06-03T03:10:43Z","snapshot_observed_at":"2026-08-04T10:15:51.477475Z","submitted_at":"2025-10-13T09:26:47Z","title":"Aligning Deep Implicit Preferences by Learning to Reason Defensively","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T08:11:03.152989Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2510.11194"},"observation_digest":"sha256:7d64de2e0a17afd9fd5a78e98a19b6f5abcc0c97d606c1bb6fa1becc5713261b","observation_id":"adc5df69-5d9b-4765-8e74-6c377448726c","resolution":{"observed_at":"2026-05-18T08:11:06.730794Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T10:15:53.984725Z","title":"LLM-as-a-judge","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2510.11194","last_updated":"2026-06-03T03:10:43Z","snapshot_observed_at":"2026-08-04T10:15:51.477475Z","submitted_at":"2025-10-13T09:26:47Z","title":"Aligning Deep Implicit Preferences by Learning to Reason Defensively","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-04T10:15:53.984725Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2510.11194"},"observation_digest":"sha256:dbfa57351a82604f5312440c203155ed85486f1b619f1bd67a5e7b8aea90bb7a","observation_id":"c8ae59d6-fcb6-49d6-ade9-2f0e63befebf","resolution":{"observed_at":"2026-08-04T10:15:53.984725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2510.13907","last_updated":"2026-04-09T04:21:03Z","snapshot_observed_at":"2026-07-06T22:32:47.087967Z","submitted_at":"2025-10-14T22:23:08Z","title":"LLM Prompt Duel Optimizer: Efficient Label-Free Prompt Optimization","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-18T07:01:43.510924Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2510.13907"},"observation_digest":"sha256:bd4b5fcc2c2cb18b9c5454b089324fc3c049ea36fe3d8ec5b214adf10ac8e92b","observation_id":"8908a78f-6480-4ed7-8c6a-940a89cf5e6c","resolution":{"observed_at":"2026-05-18T07:02:26.585565Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2511.01008","last_updated":"2026-05-10T03:37:31Z","snapshot_observed_at":"2026-07-06T22:34:43.577332Z","submitted_at":"2025-11-02T16:55:30Z","title":"MARS-SQL: A multi-agent reinforcement learning framework for Text-to-SQL","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-18T01:31:40.920567Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2511.01008"},"observation_digest":"sha256:0c24c0a47f52fac09c863f952cf82598bac35c4a9209b65cd77f624ff13a2ad2","observation_id":"1e412267-22f4-4a34-ae49-cb608c54c2aa","resolution":{"observed_at":"2026-05-18T01:32:17.346126Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2511.02356","last_updated":"2026-04-20T12:25:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-04T08:24:22Z","title":"ASTRA: An Automated Framework for Strategy Discovery, Retrieval, and Evolution for Jailbreaking LLMs","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-18T01:54:22.995178Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2511.02356"},"observation_digest":"sha256:0875c2fa1b588957d8a90c074ce5392ed0bbee07a146782c6f4bf79221b33f37","observation_id":"40d205e2-e6c5-4a37-b3a2-f9ed15946c10","resolution":{"observed_at":"2026-05-18T01:55:38.035924Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2511.03056","last_updated":"2026-04-16T17:48:38Z","snapshot_observed_at":"2026-07-06T22:34:53.346702Z","submitted_at":"2025-11-04T22:53:57Z","title":"Reading Between the Lines: The One-Sided Conversation Problem","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-18T00:39:22.598660Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2511.03056"},"observation_digest":"sha256:40cb16b26fc37d1f5202ee9dd69245ebf17b7c36762e688017efb2d428cbec76","observation_id":"b4391a83-ad5a-4391-8bc0-acb6d360c5b2","resolution":{"observed_at":"2026-05-18T00:40:33.655532Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2511.08484","last_updated":"2026-04-27T17:07:15Z","snapshot_observed_at":"2026-07-06T22:35:31.750339Z","submitted_at":"2025-11-11T17:25:44Z","title":"Patching LLM Like Software: A Lightweight Method for Improving Safety Policy in Large Language Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T23:24:40.372674Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2511.08484"},"observation_digest":"sha256:05de052b9a3f0c609aa0905be2c18738a7a21b67d0a276e62a89bee3e43eccf0","observation_id":"4d09e5d7-c31a-411a-aced-dedf4d0eb6bb","resolution":{"observed_at":"2026-05-17T23:25:28.447604Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2511.08484","last_updated":"2026-04-27T17:07:15Z","snapshot_observed_at":"2026-07-06T22:35:31.750339Z","submitted_at":"2025-11-11T17:25:44Z","title":"Patching LLM Like Software: A Lightweight Method for Improving Safety Policy in Large Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T23:24:40.372674Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2511.08484"},"observation_digest":"sha256:11b533d7f8b08aac0d2f9602ced5eeb529da0994e71591ddb4ae8cc1b1c92448","observation_id":"1ec81eda-673f-4822-9042-aa18cec22a17","resolution":{"observed_at":"2026-05-17T23:25:28.521910Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-03T22:23:30.982151Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena.Advances in Neural Information Processing Systems, 36:46595–46623, 2023a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.10868","last_updated":"2026-05-29T03:47:16Z","snapshot_observed_at":"2026-08-03T22:23:28.105321Z","submitted_at":"2025-11-14T00:35:00Z","title":"Go-UT-Bench: A Fine-Tuning Dataset for LLM-Based Unit Test Generation in Go","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T22:23:30.982151Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2511.10868"},"observation_digest":"sha256:32a2f07c10a433ddf4ba50fadb313335240f4a876dfd90219d878d09ec5eed77","observation_id":"1469f486-4021-4d8b-be53-a1ccd05ff185","resolution":{"observed_at":"2026-08-03T22:23:30.982151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-03T20:01:29.973988Z","title":"arXiv:2306.05685 [cs]","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.21522","last_updated":"2026-06-22T07:20:43Z","snapshot_observed_at":"2026-08-03T20:01:28.106236Z","submitted_at":"2025-11-26T15:52:52Z","title":"Pessimistic Verification for Open Ended Math Questions","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T20:01:29.973988Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2511.21522"},"observation_digest":"sha256:5a62df75b1db763949fcc36584cc626c1d12b9285a222d50eaa11bfccb5c97b8","observation_id":"497b1f07-4f75-4ba9-b1b0-8afabefe53b2","resolution":{"observed_at":"2026-08-03T20:01:29.973988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2512.02764","last_updated":"2026-05-12T18:50:51Z","snapshot_observed_at":"2026-07-30T07:30:39.481921Z","submitted_at":"2025-12-02T13:44:41Z","title":"PEFT-Factory: Unified Parameter-Efficient Fine-Tuning of Autoregressive Large Language Models","version":3},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-17T02:38:11.118057Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2512.02764"},"observation_digest":"sha256:9ca470667f431b4d7886e05ed0fea7e70f59c6c4d241c99399b0d250d31b70a5","observation_id":"c00e22b8-cfd9-4563-be13-59d5f11344b9","resolution":{"observed_at":"2026-05-17T02:38:53.851741Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-03T17:15:48.221901Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2512.10234","last_updated":"2026-05-30T12:47:52Z","snapshot_observed_at":"2026-08-03T17:15:46.934164Z","submitted_at":"2025-12-11T02:41:14Z","title":"InFerActive: Interactive Tree-Based Exploration of LLM Sampling for Safety Evaluation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-03T17:15:48.221901Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2512.10234"},"observation_digest":"sha256:d73f3ea841f38b32c5301aada7080c74f2da1c07eb9124abc2a26a42d8967671","observation_id":"32d189e4-7b3f-46f3-b09b-1801d253ebbf","resolution":{"observed_at":"2026-08-03T17:15:48.221901Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2512.14554","last_updated":"2026-04-17T10:30:55Z","snapshot_observed_at":"2026-08-04T02:33:48.320135Z","submitted_at":"2025-12-16T16:28:32Z","title":"VLegal-Bench: Cognitively Grounded Benchmark for Vietnamese Legal Reasoning of Large Language Models","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T21:36:24.376401Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2512.14554"},"observation_digest":"sha256:3980be458fb30ef9934effeb5de675cdfe0d70cfb378ac99b303a5360edaefd3","observation_id":"5bd1853e-0f8b-4e3f-979b-c5a6cc8abc97","resolution":{"observed_at":"2026-05-16T21:38:34.182199Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2601.00290","last_updated":"2026-04-02T21:02:45Z","snapshot_observed_at":"2026-07-06T22:40:31.869391Z","submitted_at":"2026-01-01T10:11:58Z","title":"ClinicalReTrial: Clinical Trial Redesign with Self-Evolving Agents","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T18:09:30.291505Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2601.00290"},"observation_digest":"sha256:971607f8e92fcd626c0d3b8b2f853283847c7b3b6290b353ea940ee0f4d6c47e","observation_id":"d67378b5-9685-4d9d-995c-1501f041aa9e","resolution":{"observed_at":"2026-05-16T18:11:09.389053Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2601.09536","last_updated":"2026-04-18T07:44:28Z","snapshot_observed_at":"2026-07-06T22:41:43.594456Z","submitted_at":"2026-01-14T14:57:33Z","title":"Omni-R1: Towards the Unified Generative Paradigm for Multimodal Reasoning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T14:37:05.402850Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2601.09536"},"observation_digest":"sha256:478e50606832d04e2ada56f0f1b1f2e59e24534dd4fb1cb0da695bb1d94618c0","observation_id":"2cefaacd-a7aa-457a-a0a3-e5ec2a3270a2","resolution":{"observed_at":"2026-05-16T14:37:59.991191Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2601.14053","last_updated":"2026-04-16T15:26:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-20T15:06:19Z","title":"LLMOrbit: A Circular Taxonomy of Large Language Models -From Scaling Walls to Agentic AI Systems","version":2},"reference_index":173,"source":"pdf_text","source_observed_at":"2026-05-16T12:47:28.248540Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2601.14053"},"observation_digest":"sha256:651601739140a62879cfa72d0161c80af2df556a146e525e5f8e699fbddaa820","observation_id":"93e96a40-a3d7-4f6e-ad10-243df9e1d880","resolution":{"observed_at":"2026-05-16T12:47:53.746742Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-03T08:15:37.982801Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.17717","last_updated":"2026-06-09T20:25:14Z","snapshot_observed_at":"2026-08-03T08:15:07.553014Z","submitted_at":"2026-01-25T06:40:25Z","title":"A Survey on Evaluating Quality and Trustworthiness in LLM-Generated Data","version":3},"reference_index":277,"source":"arxiv_source","source_observed_at":"2026-08-03T08:15:37.982801Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2601.17717"},"observation_digest":"sha256:bbf3f3b6968e54a94eee343a999e752d410378c4da319de2a602d02d09396b9b","observation_id":"d60f9d61-758e-48e0-9c71-f23dc7a8c5d1","resolution":{"observed_at":"2026-08-03T08:15:37.982801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-03T06:46:26.366420Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena.arXiv preprint arXiv:2306.05685, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.22025","last_updated":"2026-06-09T23:57:32Z","snapshot_observed_at":"2026-08-03T06:46:25.129139Z","submitted_at":"2026-01-29T17:32:34Z","title":"When Generic Prompt Improvements Hurt: Evaluation-Driven Iteration for LLM Applications","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T06:46:26.366420Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2601.22025"},"observation_digest":"sha256:8f4a9d63c6ca1bcb748ec61528ff20a8f55a38de83609e5b906cc04e5ed61877","observation_id":"e028793b-e67c-49a0-aeb8-c55dc563c994","resolution":{"observed_at":"2026-08-03T06:46:26.366420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-03T06:48:33.095817Z","title":"Judging LLM -as-a-judge with MT-Bench and Chatbot Arena","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.22136","last_updated":"2026-07-07T06:58:47Z","snapshot_observed_at":"2026-08-03T15:20:06.855520Z","submitted_at":"2026-01-29T18:55:46Z","title":"StepShield: When, Not Whether to Intervene on Rogue Agents","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-03T06:48:33.095817Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2601.22136"},"observation_digest":"sha256:c77ce8da63d5dbf4bb66c1df558a1f3c7807f2947711eeb1546ba1e7c44256d0","observation_id":"c33a4aee-cdf4-4ea5-9312-67612fb98833","resolution":{"observed_at":"2026-08-03T06:48:33.095817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T06:18:36.946972Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.01347","last_updated":"2026-08-03T13:57:33Z","snapshot_observed_at":"2026-08-05T08:28:18.235804Z","submitted_at":"2026-02-01T17:32:04Z","title":"A clinically validated framework for auditing AI chatbot behavior in mental health interactions","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-04T06:18:36.946972Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2602.01347"},"observation_digest":"sha256:a6b90de6f9c9a2e79d88fabb6f4a242fb37fe355e28318aa8a267b2b8acd82d8","observation_id":"b482c4bc-182e-4f88-ac1d-f7eb0691517b","resolution":{"observed_at":"2026-08-04T06:18:36.946972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-03T03:31:57.159932Z","title":"Xing, Hao Zhang, Preprint, 2025, Preprint Le, Lu, and Stern et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.07840","last_updated":"2026-06-10T04:03:01Z","snapshot_observed_at":"2026-08-03T03:31:55.760958Z","submitted_at":"2026-02-08T06:42:50Z","title":"SAGE: Scalable AI Governance & Evaluation","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-03T03:31:57.159932Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2602.07840"},"observation_digest":"sha256:7339d43a81a30075bbff10bc43af70102733d32ce34aee738753cc11ccd8cad6","observation_id":"b2e287a0-cf16-40a7-a001-24d90415dbda","resolution":{"observed_at":"2026-08-03T03:31:57.159932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2602.08819","last_updated":"2026-05-20T05:40:17Z","snapshot_observed_at":"2026-08-02T21:35:47.306662Z","submitted_at":"2026-02-09T15:55:56Z","title":"Bayesian Preference Learning for Test-Time Steerable Reward Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2602.08819"},"observation_digest":"sha256:5e1b9cf191f4a6b7a2a69afb76fb2ff47f8960d06c105acbf307d6dfb960ebdb","observation_id":"bdcf0897-057f-4d0e-8a6c-358ac17044e2","resolution":{"observed_at":"2026-05-21T13:20:10.373530Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2602.11157","last_updated":"2025-12-08T06:48:17Z","snapshot_observed_at":"2026-08-04T23:09:56.365393Z","submitted_at":"2025-12-08T06:48:17Z","title":"Response-Based Knowledge Distillation for Multilingual Jailbreak Prevention Unwittingly Compromises Safety","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-17T01:27:16.967080Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2602.11157"},"observation_digest":"sha256:2788746403629e22a2e452bf54d7f652bf9bc58ee8f363275dfe49978871b47a","observation_id":"fc65662b-f03e-46ea-a57f-27068119551c","resolution":{"observed_at":"2026-05-17T01:28:48.779663Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-02T22:43:03.629247Z","title":"Xing, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.16111","last_updated":"2026-07-29T14:24:19Z","snapshot_observed_at":"2026-08-04T13:14:41.020841Z","submitted_at":"2026-02-18T00:45:46Z","title":"Calibrate Globally, Measure Everywhere: Scaling LLM-Based Prevalence Measurement Across A/B Experiments","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T22:43:03.629247Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2602.16111"},"observation_digest":"sha256:397291e6e17d0e45d8ca4643788e3cb3c64ede63b926bf06788e0089d1c75f21","observation_id":"5999a959-c4eb-4ff2-9082-9569ed568d9e","resolution":{"observed_at":"2026-08-02T22:43:03.629247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-02T20:27:49.650865Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.23234","last_updated":"2026-06-07T00:45:24Z","snapshot_observed_at":"2026-08-03T04:32:35.932734Z","submitted_at":"2026-02-26T17:11:26Z","title":"Scaling Search Relevance: Augmenting App Store Ranking with LLM-Generated Judgments","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T20:27:49.650865Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2602.23234"},"observation_digest":"sha256:d6d685b4ccc5580c9136be0a40bf837f1f1347fd4ee26131647355b32fcab663","observation_id":"367dc6c0-0192-4d2e-983b-0a93212f9910","resolution":{"observed_at":"2026-08-02T20:27:49.650865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2603.06610","last_updated":"2026-05-22T08:27:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-19T09:46:24Z","title":"CapTrack: Multifaceted Evaluation of Forgetting in LLM Post-Training","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.06610"},"observation_digest":"sha256:5e2c441dec4e0652705d8a5f36a38dd203fd9cf17b0bdd9a9b7d004d6960eb75","observation_id":"ed96ed93-6686-4047-ad8d-e2295fafd973","resolution":{"observed_at":"2026-05-25T06:45:25.636294Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T05:54:53.441547Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.08262","last_updated":"2026-08-01T16:26:47Z","snapshot_observed_at":"2026-08-05T08:18:09.786961Z","submitted_at":"2026-03-09T11:33:05Z","title":"FinToolBench: Evaluating LLM Agents for Real-World Financial Tool Use","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T05:54:53.441547Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.08262"},"observation_digest":"sha256:3638e43e93db754aacac75dbc630f44717d0f25cb612423e0efb0355574be239","observation_id":"5064c39e-4149-440d-ae01-72fc7c1b281c","resolution":{"observed_at":"2026-08-04T05:54:53.441547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-02T18:14:54.419502Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.14473","last_updated":"2026-07-15T10:22:14Z","snapshot_observed_at":"2026-08-03T06:06:47.089700Z","submitted_at":"2026-03-15T16:31:51Z","title":"AI Can Learn Scientific Taste","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-02T18:14:54.419502Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.14473"},"observation_digest":"sha256:b96535db84eef08d63d43839998fbd5a3b587d190174ce454956fe6be9961000","observation_id":"3aaee713-1b19-4cdd-bdad-d1e9b66717db","resolution":{"observed_at":"2026-08-02T18:14:54.419502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-07-13T23:27:58.971096Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.16859","last_updated":"2026-07-01T09:16:52Z","snapshot_observed_at":"2026-07-13T23:27:58.053645Z","submitted_at":"2026-03-17T17:58:44Z","title":"SocialOmni: Benchmarking Audio-Visual Social Interactivity in Omni Models","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-07-13T23:27:58.971096Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.16859"},"observation_digest":"sha256:b0909e5e3ff2df32c9799a3cd3787fd8a6ed31e134b089238d5104226a744fdd","observation_id":"d889f58f-e4d3-4d8f-a0e0-0e0f5390f830","resolution":{"observed_at":"2026-07-13T23:27:58.971096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2603.18221","last_updated":"2026-07-26T16:27:39Z","snapshot_observed_at":"2026-08-02T18:00:38.331125Z","submitted_at":"2026-03-18T19:09:06Z","title":"Scalable and Personalized Oral Assessments Using Voice AI","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T10:05:30.015226Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.18221"},"observation_digest":"sha256:abdbc8c0194af184cd85ee613bb6d7f9b2baa90a51feffc7dcba43c7b409fac4","observation_id":"58a92dfe-273a-48f3-b703-0bea38e7b0ab","resolution":{"observed_at":"2026-05-21T10:09:59.559097Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-02T18:00:41.587603Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.18221","last_updated":"2026-07-26T16:27:39Z","snapshot_observed_at":"2026-08-02T18:00:38.331125Z","submitted_at":"2026-03-18T19:09:06Z","title":"Scalable and Personalized Oral Assessments Using Voice AI","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T18:00:41.587603Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.18221"},"observation_digest":"sha256:223deaac90f2a155d6084258727e6f733262d0a353947bbf3873774513e53e6d","observation_id":"d0a9f836-ec31-4153-9cbc-329abce6a5d1","resolution":{"observed_at":"2026-08-02T18:00:41.587603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2603.18361","last_updated":"2026-04-18T11:58:01Z","snapshot_observed_at":"2026-07-06T22:49:37.944352Z","submitted_at":"2026-03-18T23:58:37Z","title":"Synthetic Data Generation for Training Diversified Commonsense Reasoning Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T08:11:22.897371Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.18361"},"observation_digest":"sha256:71d69fed21417401f488dcd153083c51b1fc86a977b0fea707b7bd87577f3641","observation_id":"17103961-9333-406d-bad5-4f3a69ef99b8","resolution":{"observed_at":"2026-05-15T08:15:16.934625Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2603.18916","last_updated":"2026-04-12T07:45:59Z","snapshot_observed_at":"2026-07-06T22:49:42.664225Z","submitted_at":"2026-03-19T13:52:17Z","title":"Agentic Business Process Management: A Research Manifesto","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-15T08:24:56.763438Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.18916"},"observation_digest":"sha256:30ea7a695f59fee2054bef42d4b3dc0e3441080256dc856a9e8e5f8f17f95563","observation_id":"eb9902cf-92fd-48da-bec6-939bd008ded6","resolution":{"observed_at":"2026-05-15T08:25:18.154337Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-02T17:52:54.820035Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.20324","last_updated":"2026-07-21T00:19:03Z","snapshot_observed_at":"2026-08-05T03:53:20.361286Z","submitted_at":"2026-03-20T00:50:53Z","title":"When Agents Disagree: The Selection Bottleneck in Multi-Agent LLM Pipelines","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T17:52:54.820035Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.20324"},"observation_digest":"sha256:049745d0c1645014cfb5f07ccaa0abecccc5d7bd18f3b4ff3ac879ce21e80102","observation_id":"fbef79b8-acef-4017-8f48-719f9938cd32","resolution":{"observed_at":"2026-08-02T17:52:54.820035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-04T05:42:17.786502Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.28568","last_updated":"2026-08-03T06:37:39Z","snapshot_observed_at":"2026-08-05T08:25:01.149196Z","submitted_at":"2026-03-30T15:24:34Z","title":"XSPA: Crafting Imperceptible X-Shaped Sparse Adversarial Perturbations for Transferable Attacks on VLMs","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T05:42:17.786502Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2603.28568"},"observation_digest":"sha256:8c63760f1be19495032c6012e68872f2c95e4576afd21076d4a6fd5c246a83f7","observation_id":"bdfec1b5-24c0-4f2e-8e50-81691499ed1c","resolution":{"observed_at":"2026-08-04T05:42:17.786502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2604.02406","last_updated":"2026-05-16T00:42:12Z","snapshot_observed_at":"2026-08-02T21:03:26.031924Z","submitted_at":"2026-04-02T17:17:12Z","title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","version":1},"reference_index":133,"source":"pdf_text","source_observed_at":"2026-05-13T20:59:52.448832Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2604.02406"},"observation_digest":"sha256:3a8813c453817ca7837ea405978e7793daec2727a9e035c17870f4d284935853","observation_id":"8474c254-e454-4893-aec2-e0185d31bd06","resolution":{"observed_at":"2026-05-13T21:03:20.042338Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2604.02406","last_updated":"2026-05-16T00:42:12Z","snapshot_observed_at":"2026-08-02T21:03:26.031924Z","submitted_at":"2026-04-02T17:17:12Z","title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","version":2},"reference_index":133,"source":"pdf_text","source_observed_at":"2026-05-21T10:35:39.269869Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2604.02406"},"observation_digest":"sha256:45215b66ce2625ee704ccfc7693d7746fdb463ae36bd8f6aac06be6c72b2e879","observation_id":"31db940e-5b6c-4a30-95d8-e0713fadf8a6","resolution":{"observed_at":"2026-05-21T10:40:00.850281Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2306.05685/citation-record","integrity":"/paper/2306.05685/integrity","json":"/paper/2306.05685/citation-record.json","paper":"/paper/2306.05685"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.10403","last_updated":"2023-09-13T20:35:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T17:46:53Z","title":"PaLM 2 Technical Report","version":3},"cited_work":{"arxiv_id":"2305.10403","doi":"10.48550/arxiv.2305.10403","metadata_source":"pith","pith_arxiv_id":"2305.10403","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PaLM 2 Technical Report","venue":"cs.CL","work_id":"905ee9a7-ea61-4a94-bd62-2600cbe3e315","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2305.10403","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:3f25e4c3671d142ce3749bb25b176fc4b1d1819a823563d389c2219bbb706ab9","observation_id":"c3f9150a-0bf5-48f7-88b3-e3e822ae231e","resolution":{"observed_at":"2026-05-12T11:59:27.667882Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":"2204.05862","doi":"10.1016/j.respol.2005.01.014","metadata_source":"pith","pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","venue":"cs.CL","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:dca161edb57efb16ed6f8d2f2fd82c15726c6be157f30e19db4af7ec9b564670","observation_id":"7c33d917-af12-4745-be3b-a0054a21a150","resolution":{"observed_at":"2026-05-10T18:52:59.063023Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Position bias in multiple-choice questions","venue":null,"work_id":"7f6dbe66-b9f1-4ae3-a965-77f3e2ee07c2","year":1984},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:77c2609fdf756549afdb585d91a05923ac267b53008d5d27d081c8e2612f9532","observation_id":"f56df5af-05f3-4af8-bf15-aeff97581859","resolution":{"observed_at":"2026-05-10T18:52:59.181414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluations of self and others: Self-enhancement biases in social judgments","venue":null,"work_id":"7ad883c8-6967-4899-988c-5b0239fc695f","year":1986},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:0dc60a8e14de25bccf2d88282e1ebac8f6b1572ae0c9bf7f803dc2dfcb2041a8","observation_id":"a87b7471-459c-422a-9790-db258d8c2609","resolution":{"observed_at":"2026-05-10T18:52:59.184039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-03T04:49:15.195814Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":"2303.12712","doi":"10.48550/arxiv.2303.12712","metadata_source":"pith","pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","venue":"cs.CL","work_id":"a23cfe92-7f7c-424b-98d4-b386a83002fb","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:422ceeaa93a5e7e6c5449afc694727684528b342dca32468739609be1e6e4198","observation_id":"09baee0a-afe2-4161-8531-0431dc543526","resolution":{"observed_at":"2026-05-10T19:44:34.276011Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:506363e4bfc0e74626a17a7c552906c58adcdf90c3719f48854266f463003f70","observation_id":"a4505964-88d7-4bcd-8a49-358f75ddea94","resolution":{"observed_at":"2026-05-10T18:52:59.155000Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.01937","last_updated":"2023-05-03T07:28:50Z","snapshot_observed_at":"2026-08-02T17:07:32.299595Z","submitted_at":"2023-05-03T07:28:50Z","title":"Can Large Language Models Be an Alternative to Human Evaluations?","version":1},"cited_work":{"arxiv_id":"2305.01937","doi":"10.48550/arxiv.2305.01937","metadata_source":"pith","pith_arxiv_id":"2305.01937","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can large language models be an alternative to human evaluations?","venue":"cs.CL","work_id":"0fb62126-21b0-4011-8072-2f3c760d9669","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2305.01937","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:610ef263ad12f8c13d4083a7d119e6b65fff4fa5092f715d9e1080462b748513","observation_id":"814939cf-b7f6-4969-91c5-9e1bec2f57ff","resolution":{"observed_at":"2026-05-10T18:52:59.159832Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gonzalez, Ion Stoica, and Eric P","venue":null,"work_id":"6e5e2a66-cdab-4f78-b8aa-4a697085ff04","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:f96a56d2eeca7a0bb1b88934ff00ade16943a86ae263a03df2a272572821ff78","observation_id":"0d93a60b-5759-4767-8d3a-d98958c459d9","resolution":{"observed_at":"2026-05-10T18:52:59.194175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":"1803.05457","doi":"10.1162/tacl_a_00448.https://aclanthology.org/2022.tacl-1.5","metadata_source":"pith","pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","venue":"cs.AI","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","year":2018},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:f82e622a5c69a9bcd858bc4bb17c38690c13dc0282649d1f0ab4a5eff35b3036","observation_id":"9ab99aab-cafc-481c-bd8d-885eafaa8def","resolution":{"observed_at":"2026-05-10T18:52:59.163477Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-04T15:46:25.710484Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:afd787fe221155f0af8c8a854e2069abab7da137eb93743dacdf734570bc00ff","observation_id":"9f2f3665-ceb6-4f9b-8182-c9f2431e286e","resolution":{"observed_at":"2026-05-10T18:52:59.167028Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flashattention: Fast and memory-efficient exact attention with io-awareness.Advances in Neural Information Processing Systems, 35:16344–16359","venue":null,"work_id":"3e33869a-9182-4187-90f6-0a1149bdbb50","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:bed64c4532298f7b56b9e4c79d2cecd0dc9f949176cfcca5b654e6dfab59df97","observation_id":"7b6674a3-fecb-4a1a-ac7d-254840df5b89","resolution":{"observed_at":"2026-05-10T18:52:59.202270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14314","last_updated":"2023-05-23T17:50:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T17:50:33Z","title":"QLoRA: Efficient Finetuning of Quantized LLMs","version":1},"cited_work":{"arxiv_id":"2305.14314","doi":"10.1007/978-3-031-86644-9","metadata_source":"pith","pith_arxiv_id":"2305.14314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"QLoRA: Efficient Finetuning of Quantized LLMs","venue":"cs.LG","work_id":"d3fdf68e-3a5e-48b5-8a18-7a9137479c55","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2305.14314","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:253a3d97659bfa9b5771152cf643c1b0b8de6f1625bc6248febd78585bd1ee07","observation_id":"60135306-7518-4a01-a77c-02b859d36201","resolution":{"observed_at":"2026-05-11T13:29:54.058781Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.12420","last_updated":"2024-05-05T13:13:02Z","snapshot_observed_at":"2026-07-06T15:45:13.905037Z","submitted_at":"2023-06-21T17:58:25Z","title":"LMFlow: An Extensible Toolkit for Finetuning and Inference of Large Foundation Models","version":2},"cited_work":{"arxiv_id":"2306.12420","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.12420","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lmflow: An extensible toolkit for finetuning and inference of large foundation models","venue":null,"work_id":"a1130a23-fe86-47b0-b557-7359588ceaa2","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2306.12420","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:5f415d22351b3b3c924f002368fadb18c4cc93b84719b9576751e73cb58a7865","observation_id":"5e41114d-4857-432e-9fe4-75e731cf2093","resolution":{"observed_at":"2026-05-10T18:52:59.174675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14387","last_updated":"2024-01-08T04:46:56Z","snapshot_observed_at":"2026-07-06T15:31:51.249701Z","submitted_at":"2023-05-22T17:55:50Z","title":"AlpacaFarm: A Simulation Framework for Methods that Learn from Human Feedback","version":4},"cited_work":{"arxiv_id":"2305.14387","doi":"10.48550/arxiv.2305.14387","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.14387","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Yann Dubois, Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani, Jimmy Ba, Carlos Guestrin, Percy Liang, and Tatsunori B Hashimoto","venue":"arXiv (Cornell University)","work_id":"a875adf2-8826-466d-bf52-896ee15632ea","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2305.14387","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:565e1a5790ebbb763132b6c72233119f47f55d21fcf500630e14ee80a231adcd","observation_id":"41c776e8-d9d2-4e35-a3a2-e329d193b2fa","resolution":{"observed_at":"2026-05-10T18:52:59.178783Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.05719","last_updated":"2022-12-21T08:12:46Z","snapshot_observed_at":"2026-07-06T14:16:48.496031Z","submitted_at":"2022-11-10T17:37:04Z","title":"MMDialog: A Large-scale Multi-turn Dialogue Dataset Towards Multi-modal Open-domain Conversation","version":3},"cited_work":{"arxiv_id":"2211.05719","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2211.05719","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mmdialog: A large-scale multi-turn dialogue dataset towards multi-modal open-domain conversation","venue":null,"work_id":"084e2818-89c1-4802-b493-2ae54668ce4d","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2211.05719","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:eaaa4ff5e450e0851169eeecf7b7e3e199161ec5b6d9b692f4d1253be0c701c6","observation_id":"a702fc8b-546c-4aac-a4d7-b38999c4ffe1","resolution":{"observed_at":"2026-05-10T18:52:59.067341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Koala: A dialogue model for academic research","venue":null,"work_id":"3fc02f21-2595-4d25-aad8-30ee3a513113","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:3809ff49a973f803fa58fbfb82200719984c13dc447c3325122f21370b929963","observation_id":"bf0492e9-4aa1-481f-b6ab-0344a36a23cc","resolution":{"observed_at":"2026-05-10T18:52:59.215697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.15056","last_updated":"2023-07-19T14:10:55Z","snapshot_observed_at":"2026-08-01T16:32:36.103733Z","submitted_at":"2023-03-27T09:59:48Z","title":"ChatGPT Outperforms Crowd-Workers for Text-Annotation Tasks","version":2},"cited_work":{"arxiv_id":"2303.15056","doi":"10.48550/arxiv.2303.15056","metadata_source":"arxiv_reference","pith_arxiv_id":"2303.15056","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2303.15056 , year=","venue":"arXiv (Cornell University)","work_id":"9696432b-883b-4fa9-9de5-2f384b96f5c4","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2303.15056","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:bca8885e0a0ee38b3f0799442ccec1786cc368000d816e59d4df8deba7f217d2","observation_id":"882919d0-3361-467d-8025-cc4e9587a3b4","resolution":{"observed_at":"2026-05-10T18:52:59.071717Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15717","last_updated":"2023-05-25T05:00:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-25T05:00:12Z","title":"The False Promise of Imitating Proprietary LLMs","version":1},"cited_work":{"arxiv_id":"2305.15717","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.15717","snapshot_observed_at":"2026-07-04T05:09:36.906364Z","title":"The False Promise of Imitating Proprietary LLMs","venue":"cs.CL","work_id":"f843d86a-7fd6-4b0d-bf8b-f1ad3fbc3f04","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2305.15717","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:eb41c134367dc74353e72138aaa126ce7ac35d0e97288bf4f858f84cfc25fb7a","observation_id":"d19eb9a3-a7a6-4d82-b093-2fd66a707a2c","resolution":{"observed_at":"2026-05-18T06:54:31.902059Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":"2009.03300","doi":"10.48550/arxiv.2009.03300","metadata_source":"pith","pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Massive Multitask Language Understanding","venue":"cs.CY","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","year":2020},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:07d375ccdf66496b6e46ec22eb02f6e01d4e97b59ae2b6d92662d8815a47a83f","observation_id":"92c0b87e-7c8c-425d-adda-f0ffa9fccc7b","resolution":{"observed_at":"2026-05-10T18:52:59.079924Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.07736","last_updated":"2023-03-15T19:16:45Z","snapshot_observed_at":"2026-07-06T14:52:07.793181Z","submitted_at":"2023-02-11T03:13:54Z","title":"Is ChatGPT better than Human Annotators? Potential and Limitations of ChatGPT in Explaining Implicit Hate Speech","version":2},"cited_work":{"arxiv_id":"2302.07736","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2302.07736","snapshot_observed_at":"2026-07-04T13:29:51.132766Z","title":"Is chatgpt better than human annotators? potential and limitations of chatgpt in explaining implicit hate speech","venue":null,"work_id":"b5fe86c8-0a13-43db-83c8-abb1e75d3995","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2302.07736","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:1991ae650a8783d0b7265ab7f4d72cb35ad076ef8605bc30556ecb9aa1e460a7","observation_id":"276d5d61-adce-4dee-a3d2-4613ec4588f7","resolution":{"observed_at":"2026-05-10T18:52:59.083672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dynabench: Rethinking benchmarking in nlp","venue":null,"work_id":"b144e55a-99bb-44df-a541-1eedc254eda1","year":2021},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:90eabb4c63e392e11c4fde42785e97244cee6e83b4fc576b91bc728e0661d9b2","observation_id":"2ef3ec40-fe00-42b7-8282-502c13017f18","resolution":{"observed_at":"2026-05-10T18:52:59.233936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.14602","last_updated":"2021-03-08T15:09:45Z","snapshot_observed_at":"2026-08-03T16:42:08.411253Z","submitted_at":"2020-04-30T06:25:16Z","title":"Look at the First Sentence: Position Bias in Question Answering","version":4},"cited_work":{"arxiv_id":"2004.14602","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2004.14602","snapshot_observed_at":"2026-07-04T08:49:41.497818Z","title":"Look at the first sentence: Position bias in question answering","venue":null,"work_id":"394f1cca-a21b-4a46-bff0-175f89f14682","year":2004},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2004.14602","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:1a161f9ce8ce07dec8a04b95ee350e22e3edf51d4495eb6f671f6123633d8d54","observation_id":"de397f39-9f19-4e9f-bcd2-2b16d88d5019","resolution":{"observed_at":"2026-05-10T18:52:59.087290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07327","last_updated":"2023-10-31T11:38:07Z","snapshot_observed_at":"2026-07-06T15:15:53.807406Z","submitted_at":"2023-04-14T18:01:29Z","title":"OpenAssistant Conversations -- Democratizing Large Language Model Alignment","version":2},"cited_work":{"arxiv_id":"2304.07327","doi":"10.48550/arxiv.2304.07327","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.07327","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"R., Stevens, K., Barhoum, A., Duc, N","venue":"arXiv (Cornell University)","work_id":"34ddaebd-853b-4596-b47d-a42b9340305e","year":2024},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2304.07327","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:f638e2f4c6d04a0bd0889e9ca267e3a2e6b7d9589b9949587cbbda4657031430","observation_id":"d0b41416-d58c-4176-9f98-fb9772bc5685","resolution":{"observed_at":"2026-05-10T18:52:59.090974Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":"2211.09110","doi":"10.1007/bf01194075","metadata_source":"pith","pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Holistic Evaluation of Language Models","venue":"cs.CL","work_id":"cc02a01e-7218-47dc-8e66-3333e7e4adec","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:64de06bbe8c23b1eaf4172cef65db298adfd08776fdfb8a3f5179bcb51f7de9b","observation_id":"4c4b2457-6247-403d-a0b9-40662455db54","resolution":{"observed_at":"2026-05-10T18:52:59.095035Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T10:37:02.016439Z","title":"Rouge: A package for automatic evaluation of summaries","venue":null,"work_id":"ab9cb640-34d3-454d-8045-b23d7e537d81","year":2004},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:a017d2f816558d16de63cad6b50dfc640c65a15d3f654903120536165637e1f6","observation_id":"56ddae28-18e5-4c78-ad64-cccf5a2c424d","resolution":{"observed_at":"2026-05-10T18:52:59.189235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.07958","last_updated":"2022-05-08T02:43:02Z","snapshot_observed_at":"2026-08-03T16:26:48.747700Z","submitted_at":"2021-09-08T17:15:27Z","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","version":2},"cited_work":{"arxiv_id":"2109.07958","doi":"10.48550/arxiv.2109.07958","metadata_source":"pith","pith_arxiv_id":"2109.07958","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","venue":"cs.CL","work_id":"22e3b047-a6e8-4c4c-b62e-173b545a1a45","year":2021},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2109.07958","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:f647aa3e518ac660f52bffdce400e100bbae253549d6cdc2ad6b5dc982dfd174","observation_id":"6420962c-9527-41b1-a684-e98a96cd9d4c","resolution":{"observed_at":"2026-05-11T21:48:54.871535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-18T08:21:09.956602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T08:21:09.956602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.13688","last_updated":"2023-02-14T16:33:33Z","snapshot_observed_at":"2026-07-06T14:46:37.657675Z","submitted_at":"2023-01-31T15:03:44Z","title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","version":2},"cited_work":{"arxiv_id":"2301.13688","doi":"10.48550/arxiv.2301.13688","metadata_source":"pith","pith_arxiv_id":"2301.13688","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","venue":"cs.AI","work_id":"56b2f752-8022-4d66-983c-2bb9cb4b2b77","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2301.13688","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:001578ca19219f872950c4d99f66d6c6621f6dd4d5862832d4e6d80d5895d14a","observation_id":"5de40e0e-d688-4fbc-9ca9-36e30c1e56b6","resolution":{"observed_at":"2026-05-10T18:52:59.104420Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cross-task general- ization via natural language crowdsourcing instructions","venue":null,"work_id":"e0397d5e-a902-4d54-a040-bc457d5ab77f","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:6beb0c59b1271778fac4c3a741fdc665e300b888bdf52b44ff7b455d243d979b","observation_id":"c6f0fdbc-bd59-490a-826f-8bad7e6cd33d","resolution":{"observed_at":"2026-05-10T18:52:59.199605Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evals is a framework for evaluating llms and llm systems, and an open-source registry of benchmarks","venue":null,"work_id":"9e0d82e1-1a91-4245-adad-e982c2d70369","year":null},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:df2ceb8cdf1a0d59b5b7b6d3584fa3b0a9a3541cbb8c68b5e7646accb6f83a63","observation_id":"023175b4-86cf-413d-8ee2-5c0fdef23333","resolution":{"observed_at":"2026-05-10T18:52:59.205005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4 technical report","venue":null,"work_id":"e0243b28-1af2-4ed1-8372-1f571833c570","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:a63be6eb365abb7ec6cfd1a6455ce41eda1cacfc0f9a570974a8bc348c3a6363","observation_id":"1b604ac3-fa49-490e-80bc-e49732fbd078","resolution":{"observed_at":"2026-05-10T18:52:59.207449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"22a399d7-c4ba-4344-86b9-7885598b17ce","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:dbd63f8b996a739d2e821e22668d00d1bcf93e72669ae86ff3a9c71fea8a0f61","observation_id":"de3513db-3987-4583-b27a-f726addd9802","resolution":{"observed_at":"2026-05-10T18:52:59.210440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bleu: a method for automatic evaluation of machine translation","venue":null,"work_id":"61d34a5f-3525-4046-b217-7110c889eb7a","year":2002},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:9e54176d9e50c8fb014ca1fd4728d0129ecee5309ccaff140eca80f129bb8efc","observation_id":"1cf3c607-128d-4481-a6ee-b71b6901bd4c","resolution":{"observed_at":"2026-05-10T18:52:59.212883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":"2304.03277","doi":"10.48550/arxiv.2304.03277","metadata_source":"pith","pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Instruction Tuning with GPT-4","venue":"cs.CL","work_id":"fd515477-f9f1-48aa-9feb-a3308e7656bb","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:c444eceff92cc77a0de1514c28186d14748ae222a2e4c60c9fa25c3a2c7c13e4","observation_id":"ec220355-10e0-493f-a74f-774e3da1878d","resolution":{"observed_at":"2026-05-14T17:04:18.193148Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Center-of-inattention: Position biases in decision-making","venue":null,"work_id":"64eaa427-7bed-405e-b180-80767b066218","year":2006},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:be99067530335b556eb34fafe768843800162f1abfffd8da009c42984e1acb07","observation_id":"ef89e6ef-394c-495e-be24-673beebaf236","resolution":{"observed_at":"2026-05-10T18:52:59.225077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Coqa: A conversational question answering challenge","venue":null,"work_id":"e7740820-a3fb-47bb-92ee-64da7d0b6d97","year":2019},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:0f7ae0dacb21f4c995b932998529e01deb7a6adfe050c7522788fce797d233a2","observation_id":"e7dd84eb-7ec4-4080-bb57-e394a44036df","resolution":{"observed_at":"2026-05-10T18:52:59.228327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Winogrande: An adversarial winograd schema challenge at scale","venue":null,"work_id":"77c34b1f-a890-4ade-b1db-49d3ff74372f","year":2021},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:6f2a76970e49001b3a9ecf7557f6e944d2501768d1a94e8d2c2aa857690c6e8c","observation_id":"54fc39ce-0e59-4560-b018-97fb962b959f","resolution":{"observed_at":"2026-05-10T18:52:59.231283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":"2206.04615","doi":"10.1162/tacl_a_00688","metadata_source":"pith","pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","venue":"cs.CL","work_id":"bb63abb3-0d50-4362-b97c-b5e725b03b39","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:ecd5f8b20fba1e769c34c0170f8f9c0df8d29d1981974dbaf9f406d84d9f1ba6","observation_id":"7e1adb2e-beef-4bda-b5ac-906e746478fb","resolution":{"observed_at":"2026-05-10T23:26:25.883762Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T02:55:53.827539Z","title":"Hashimoto","venue":null,"work_id":"35104f7e-ffc3-488e-b66a-a4d3d230e657","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:6de33b53196fd85ec04bd744788bfb48ae17657db261186bcadc866de0968969","observation_id":"fc722651-daa7-4fec-b78f-d007ae6f3224","resolution":{"observed_at":"2026-05-10T18:52:59.239419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:b2030a654d95917861c58c65de622d8b520b5a42612ee758826c2eaffcb5e3e5","observation_id":"b379bdfb-6423-4660-acf0-23794153ae71","resolution":{"observed_at":"2026-05-10T18:52:59.117544Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17926","last_updated":"2023-08-30T13:22:35Z","snapshot_observed_at":"2026-08-03T19:35:11.838629Z","submitted_at":"2023-05-29T07:41:03Z","title":"Large Language Models are not Fair Evaluators","version":2},"cited_work":{"arxiv_id":"2305.17926","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.17926","snapshot_observed_at":"2026-07-11T02:27:48.661096Z","title":"Large Language Models are not Fair Evaluators","venue":"cs.CL","work_id":"d04a3326-b025-498f-a8f0-3d2df254a77f","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2305.17926","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:0ef74b023f501569e4e190640f83e19edf5835a4f955399b4eb1abdcbe3a7407","observation_id":"58e12cdb-2880-4458-a142-0261ec8e81d6","resolution":{"observed_at":"2026-05-17T12:10:42.624770Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Position bias estimation for unbiased learning to rank in personal search","venue":null,"work_id":"fe2e5aaf-22eb-4f37-8131-dc17b71ff975","year":2018},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:de98170c21577aee76ca2430598c8c61391382c4da1bb25c62f69ca0863b5b67","observation_id":"08fa3ac1-8d8b-4e67-95c4-57a59c7ebdb9","resolution":{"observed_at":"2026-05-10T18:52:59.196898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pandalm: An automatic evaluation benchmark for llm instruction tuning optimization","venue":null,"work_id":"61dcbb90-791a-4ad7-a079-089a1fed3961","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:fc37614024fb2f1359f51346de85a34d3e5a48c29b413591ae8c1f7fca53909b","observation_id":"99088c72-458b-43cf-9221-c81e0fe11a8c","resolution":{"observed_at":"2026-05-10T18:52:59.218309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.04751","last_updated":"2023-10-30T20:36:20Z","snapshot_observed_at":"2026-08-04T08:05:09.732049Z","submitted_at":"2023-06-07T19:59:23Z","title":"How Far Can Camels Go? Exploring the State of Instruction Tuning on Open Resources","version":2},"cited_work":{"arxiv_id":"2306.04751","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.04751","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2306.04751 , year=","venue":null,"work_id":"be09ac68-3260-4c2f-bfcb-1ae1b5717468","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2306.04751","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:115d1401bb3e9deeaa7a7f598f69b41410a05ee0ee05a14bf125795eedc57f57","observation_id":"c4054626-5488-4272-9046-d0d7255c4bf2","resolution":{"observed_at":"2026-05-10T18:52:59.124927Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Smith, Daniel Khashabi, and Hannaneh Hajishirzi","venue":null,"work_id":"8ea8bb68-cdf0-48e9-bb14-a12c9c3676c1","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:4b151c4542b2663b009730b63717d8016f378f9a60049100b8c7d96228130e71","observation_id":"042d7b39-f694-44d2-9384-80ecd97ac6f6","resolution":{"observed_at":"2026-05-10T18:52:59.186766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Super-naturalinstructions:generalization via declarative instructions on 1600+ tasks","venue":null,"work_id":"7b52eed2-cb51-48a0-875c-4c9d1fcc60de","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:274f4c04cc9509fd102e9a5bf36b238c713c50d6fa2f9286f092276e479d93bd","observation_id":"a8da3b96-f217-4222-963f-415b8e76bddb","resolution":{"observed_at":"2026-05-10T18:52:59.191912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.01652","last_updated":"2022-02-08T20:26:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-09-03T17:55:52Z","title":"Finetuned Language Models Are Zero-Shot Learners","version":5},"cited_work":{"arxiv_id":"2109.01652","doi":"10.3030/823782","metadata_source":"pith","pith_arxiv_id":"2109.01652","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Finetuned Language Models Are Zero-Shot Learners","venue":"cs.CL","work_id":"7ed6cdaa-ed67-4db4-aceb-b7e1b0e6e7c4","year":2021},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2109.01652","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:34e06bd31c3426c70cb93663e51b079d50955b62433b6e8885347018c81696be","observation_id":"2f529359-2d9e-40ff-b5f2-39e47f9b7ab0","resolution":{"observed_at":"2026-05-10T21:14:13.786069Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":"2201.11903","doi":"10.48550/arxiv.2201.11903","metadata_source":"pith","pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","venue":"cs.CL","work_id":"d1cf6693-a082-403c-ada9-dac7b96341f9","year":2022},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:3af49dfbbeb83ba18150ca76a5117803c16bbd8933b0316765256a40a53cedb6","observation_id":"6685e195-3cf2-4ad8-b95a-195380b67c1e","resolution":{"observed_at":"2026-05-10T18:52:59.131510Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:13.648188+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:13.648188+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12244","last_updated":"2025-05-27T06:49:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-24T16:31:06Z","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","version":3},"cited_work":{"arxiv_id":"2304.12244","doi":"10.48550/arxiv.2304.12244","metadata_source":"pith","pith_arxiv_id":"2304.12244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","venue":"cs.CL","work_id":"fc22911b-0900-4ad2-b106-337f13279965","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2304.12244","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:d4916d89185b690abacac299ef5d182935e9fc8d2f40ae7de14a58cce1667089","observation_id":"0ae0f4c0-46e2-422c-96b8-525f61e08af9","resolution":{"observed_at":"2026-05-13T07:28:25.169141Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"SkyPilot: An intercloud broker for sky computing","venue":null,"work_id":"6cf91b4f-9ea2-4e86-8205-2038caabc0ba","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:7a598c467d0212623713df63a899ac21963d96a868469a7af6cc54fb8b356072","observation_id":"ac1331a5-b308-4285-9989-f21ed41ea6ae","resolution":{"observed_at":"2026-05-10T18:52:59.236715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.07830","last_updated":"2019-05-19T23:57:23Z","snapshot_observed_at":"2026-07-31T00:09:56.948833Z","submitted_at":"2019-05-19T23:57:23Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","version":1},"cited_work":{"arxiv_id":"1905.07830","doi":"10.48550/arxiv.1905.07830","metadata_source":"pith","pith_arxiv_id":"1905.07830","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","venue":"cs.CL","work_id":"79f44c0c-96f4-4edb-bc50-a3c9d6b85936","year":2019},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/1905.07830","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:ea44e2b70fb679b67d8265bd030bfcb147ebc6ec593dcdc6066e5d23b77d443e","observation_id":"26a16eed-6774-4b15-80bd-9a99e39b7af6","resolution":{"observed_at":"2026-05-11T02:56:24.661063Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06364","last_updated":"2023-09-18T14:23:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-13T09:39:30Z","title":"AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models","version":2},"cited_work":{"arxiv_id":"2304.06364","doi":"10.48550/arxiv.2304.06364","metadata_source":"pith","pith_arxiv_id":"2304.06364","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models","venue":"cs.CL","work_id":"d42c58d5-eeb0-462a-92ab-3081ee269e59","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2304.06364","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:54f3a82a79399391e39063d02910feca8282866de40aa61b9c97cbcd3e6bc2e9","observation_id":"c1b92dad-805a-47d2-943c-dc989a98b5ff","resolution":{"observed_at":"2026-05-16T10:03:59.570579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11206","last_updated":"2023-05-18T17:45:22Z","snapshot_observed_at":"2026-07-06T15:29:24.259073Z","submitted_at":"2023-05-18T17:45:22Z","title":"LIMA: Less Is More for Alignment","version":1},"cited_work":{"arxiv_id":"2305.11206","doi":"10.48550/arxiv.2305.11206","metadata_source":"pith","pith_arxiv_id":"2305.11206","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LIMA: Less Is More for Alignment","venue":"cs.CL","work_id":"3986a077-ad54-4cba-bd8f-914fe1862115","year":2023},"citing_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T18:52:59.033645Z"},"links":{"cited_paper":"/paper/2305.11206","citing_paper":"/paper/2306.05685"},"observation_digest":"sha256:03e8f7e2338fcca10ad5e8f12a15090cadf7fe19b3963a695dccc93b945f49ad","observation_id":"fb658417-ddde-4e1e-85b2-65ebdd99d0de","resolution":{"observed_at":"2026-05-17T11:34:13.088019Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":0,"verified_exact":27,"verified_fuzzy":21},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 100 inbound Pith citation observations for arXiv:2306.05685."}