{"as_of":"2026-08-05T08:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3cab1e69729a1bbf6b56c39d4f8350765a4e4c753408d0b148bb7ce494e08cdb","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T11:50:26.030339Z","state":"measured"},{"denominator":144,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":144,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":93,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":93,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T05:31:06.543189Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":322,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T08:20:01.011625Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2312.07104"},"observation_digest":"sha256:31e1e67a0bd26d602cc776852ab8e5771d4df13b0863e27d6b8ab80c7621d1ef","observation_id":"f6666ca9-1e4f-4bdb-b702-6f659369aceb","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2405.14573","last_updated":"2025-04-06T20:37:50Z","snapshot_observed_at":"2026-07-06T18:18:39.093732Z","submitted_at":"2024-05-23T13:48:54Z","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T12:06:13.697487Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2405.14573"},"observation_digest":"sha256:3a9739f2b0be6c2d8da530de8890b618a6e3cb6cc9664563e16fa7cb483588da","observation_id":"be987ecb-0262-42e5-baaa-7a61352de344","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:3b33ade104382a1e52118a826be6d60dbb67dd5f0698ca1d0bf8e9cba02186ec","observation_id":"8a07d66b-ea87-4efa-9588-0d8643c0835f","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T23:10:40.420241Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2406.04244"},"observation_digest":"sha256:0c561f591a7955f13940b80f92db26fdc6e548fe4abf18f609749c1394b9fe6e","observation_id":"d53d4858-397a-42ed-a2d5-1225a7f1286b","resolution":{"observed_at":"2026-05-22T23:10:40.889832Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T00:05:03.547664Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2406.16860"},"observation_digest":"sha256:5f9bf2ebaf635a3841234b299dfafef81864369af017eff0770ed1dde6cc22b3","observation_id":"1c431ca0-cfef-482f-85f4-a53d9773a37c","resolution":{"observed_at":"2026-05-17T00:05:03.840660Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T04:48:26.303240Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2406.19314"},"observation_digest":"sha256:4c4986861effa19849b5436e5e396bee5c37228d8f9896ed0e9670f08a2cfd17","observation_id":"f619c2b6-d3c7-47a0-9f86-70eb839dc106","resolution":{"observed_at":"2026-05-15T04:48:26.439008Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T12:33:38.867604Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2409.12186"},"observation_digest":"sha256:131e18df4d88bb3c5f105f2f1d9461d63c85fc21fd8969275bc90fdbc7e37683","observation_id":"94c915b5-1495-4b1f-84e9-3f91c581087c","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2410.07985","last_updated":"2024-12-24T04:04:30Z","snapshot_observed_at":"2026-08-02T10:35:57.002867Z","submitted_at":"2024-10-10T14:39:33Z","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","version":3},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-15T09:09:14.884516Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2410.07985"},"observation_digest":"sha256:d27bbb2f384f5d180b05423b19d1804400f347f1ee26f2f1d79a6966bed33575","observation_id":"200f367c-3777-429e-bee8-0ac5ee91faeb","resolution":{"observed_at":"2026-05-15T09:09:15.078017Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2412.14164","last_updated":"2024-12-18T18:58:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-18T18:58:50Z","title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","version":1},"reference_index":185,"source":"arxiv_source","source_observed_at":"2026-05-17T07:51:12.953777Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2412.14164"},"observation_digest":"sha256:dea984dd7f08ec71f05faeb47ccb2b73f781c282edfc086b240ce8af0878fe1a","observation_id":"9ed17aba-e7da-470c-8e1c-0dccf8518699","resolution":{"observed_at":"2026-05-17T07:51:13.377092Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2501.06322","last_updated":"2025-01-10T19:56:50Z","snapshot_observed_at":"2026-07-30T22:07:13.869232Z","submitted_at":"2025-01-10T19:56:50Z","title":"Multi-Agent Collaboration Mechanisms: A Survey of LLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T15:54:54.146003Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2501.06322"},"observation_digest":"sha256:86f01d6a71da70e74db145f2ded29c9159e9f76920ae4a4abb7e578fc06299e5","observation_id":"8802a63b-d80b-4ce6-b2f5-eb01ac5bf8ef","resolution":{"observed_at":"2026-05-13T15:54:54.330650Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2501.09775","last_updated":"2026-05-03T15:15:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T10:27:51Z","title":"Multiple Choice Questions: Reasoning Makes Large Language Models (LLMs) More Self-Confident, Especially When They are Wrong","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-23T05:37:22.955895Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2501.09775"},"observation_digest":"sha256:bd89d4a341e9d65afe15ce935c0bbc640c3a8fc95d9213adcf5ce876e2374751","observation_id":"b32ffa91-5b0b-4bec-8a41-5755d99743f5","resolution":{"observed_at":"2026-05-23T05:37:36.068625Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2504.16155","last_updated":"2026-05-07T09:59:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:52:04Z","title":"PRIMETIME : Limits of LLMs in Temporal Primitives","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-22T18:36:48.376877Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2504.16155"},"observation_digest":"sha256:220688ac60d6e95881c09404c80bbee9b178757477537e090742ad23ff0af088","observation_id":"0fe1e8af-6e58-4e29-8e82-be82ed0ee751","resolution":{"observed_at":"2026-05-22T18:36:58.884817Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:a44637557c980bc5ffb0f72173f26c32ee8c14db95404325c9f452b95135e47d","observation_id":"5c39fabb-892a-4e64-8eb7-441385c8b786","resolution":{"observed_at":"2026-05-19T05:57:08.076312Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2507.15003","last_updated":"2025-07-20T15:15:58Z","snapshot_observed_at":"2026-07-06T21:59:59.698990Z","submitted_at":"2025-07-20T15:15:58Z","title":"The Rise of AI Teammates in Software Engineering (SE) 3.0: How Autonomous Coding Agents Are Reshaping Software Engineering","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T17:38:55.159673Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2507.15003"},"observation_digest":"sha256:8198fd79fbd1706c4f0c39ebbf589b522483058741236c294648c53e4933eeea","observation_id":"f90d5d43-c538-407f-9790-41783aabd005","resolution":{"observed_at":"2026-05-15T17:38:55.329835Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2509.00673","last_updated":"2026-05-04T01:37:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-31T03:00:55Z","title":"Confident, Calibrated, or Complicit: Safety Alignment and Ideological Bias in LLM Hate Speech Detection","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-18T20:27:34.917353Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2509.00673"},"observation_digest":"sha256:47b38913968c85c6a291b0e4b38ae8c2784fa9481644a02da442b296d393e4e9","observation_id":"04aaa514-edc1-417e-8aa5-9e22085d8c99","resolution":{"observed_at":"2026-05-18T20:31:50.077992Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T05:11:00.893392Z","title":"Zheng, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05852","last_updated":"2025-09-06T22:29:17Z","snapshot_observed_at":"2026-08-05T05:10:49.872306Z","submitted_at":"2025-09-06T22:29:17Z","title":"Fisher Random Walk: Automatic Debiasing Contextual Preference Inference for Large Language Model Evaluation","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T05:11:00.893392Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2509.05852"},"observation_digest":"sha256:b0a42e210a8402df5fe88d2ebbf1095c12675cf2579e7e54e16547f8c72e5971","observation_id":"ad5d5dcb-98de-4436-a9cd-69f0b1bebda3","resolution":{"observed_at":"2026-08-05T05:11:00.893392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T05:31:06.543189Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09707","last_updated":"2025-09-05T21:46:41Z","snapshot_observed_at":"2026-08-05T05:31:01.917086Z","submitted_at":"2025-09-05T21:46:41Z","title":"LLM-Based Instance-Driven Heuristic Bias In the Context of a Biased Random Key Genetic Algorithm","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T05:31:06.543189Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2509.09707"},"observation_digest":"sha256:8a41074fc3c79ffc85637c4857e33b408cf51a52c689781b2c8155aa536fdacd","observation_id":"2f46e505-ae9e-4166-b133-6696235b9be4","resolution":{"observed_at":"2026-08-05T05:31:06.543189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-04T17:15:27.863520Z","title":"Paul Christiano, Jan Leike, Tom B","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.11026","last_updated":"2025-09-14T01:33:14Z","snapshot_observed_at":"2026-08-04T17:15:27.499140Z","submitted_at":"2025-09-14T01:33:14Z","title":"Rethinking Human Preference Evaluation of LLM Rationales","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-04T17:15:27.863520Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2509.11026"},"observation_digest":"sha256:d28bda84d0c58bd04ec4725b68b09c084c8435a0e0ffcd0bdcee50f5787d076a","observation_id":"b9c9a3ff-c413-4d5f-845b-cc6fadf33474","resolution":{"observed_at":"2026-08-04T17:15:27.863520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2511.01101","last_updated":"2026-04-19T21:05:03Z","snapshot_observed_at":"2026-07-31T05:34:22.759829Z","submitted_at":"2025-11-02T22:33:19Z","title":"TSVer: A Benchmark for Fact Verification Against Time-Series Evidence","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-18T00:56:33.806425Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2511.01101"},"observation_digest":"sha256:679a38b9629fed348739bf442d80b9f10284056cc65f8d23c3a29979f1601598","observation_id":"0d00c505-e7a5-4e03-89aa-5c8ffbab6b7e","resolution":{"observed_at":"2026-05-18T01:00:34.071723Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2601.14053","last_updated":"2026-04-16T15:26:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-20T15:06:19Z","title":"LLMOrbit: A Circular Taxonomy of Large Language Models -From Scaling Walls to Agentic AI Systems","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T12:47:28.248540Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2601.14053"},"observation_digest":"sha256:2e0c6d6a241c748f3043b70cc000199e25f2645a8496e427958c4904d52d686e","observation_id":"20479b87-f7c6-4495-b5ad-676257f27f08","resolution":{"observed_at":"2026-05-16T12:47:53.777048Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2601.17617","last_updated":"2026-04-28T21:21:45Z","snapshot_observed_at":"2026-07-06T22:42:52.947644Z","submitted_at":"2026-01-24T22:42:43Z","title":"Agentic Search in the Wild: Intents and Trajectory Dynamics from 14M+ Real Search Requests","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T11:20:20.651794Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2601.17617"},"observation_digest":"sha256:d93cf47158725a446401dab12af8d7d00176e057c01d8e682721bcd1d2e6be3b","observation_id":"508110b7-512b-432c-9656-451603ea0949","resolution":{"observed_at":"2026-05-16T11:20:53.151926Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-02T20:24:29.500934Z","title":"Chatbot arena: An open platform for evaluating LLMs by human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.23565","last_updated":"2026-07-11T00:51:28Z","snapshot_observed_at":"2026-08-02T20:24:26.116561Z","submitted_at":"2026-02-27T00:21:36Z","title":"Dynamics of Learning under User Choice: Overspecialization and Peer-Model Probing","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T20:24:29.500934Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2602.23565"},"observation_digest":"sha256:3e28b5faf8cdd9def71722788fa288a4a8a99826a2cd9c3583c6f0d4db6d92a8","observation_id":"88bf4585-498d-41dd-a5ec-289dfb7042e9","resolution":{"observed_at":"2026-08-02T20:24:29.500934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-02T17:52:54.735172Z","title":"Gon- zalez, and Ion Stoica","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.20324","last_updated":"2026-07-21T00:19:03Z","snapshot_observed_at":"2026-08-05T03:53:20.361286Z","submitted_at":"2026-03-20T00:50:53Z","title":"When Agents Disagree: The Selection Bottleneck in Multi-Agent LLM Pipelines","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T17:52:54.735172Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2603.20324"},"observation_digest":"sha256:f213a602766ac7e3ff778fee4771b73fe52ca36c8ae0071a53804e7bdf8f2d49","observation_id":"b57c1488-86c7-4bb4-ab30-95242d27c7db","resolution":{"observed_at":"2026-08-02T17:52:54.735172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2603.24591","last_updated":"2026-04-03T16:34:38Z","snapshot_observed_at":"2026-07-30T08:49:15.136517Z","submitted_at":"2026-03-25T17:58:56Z","title":"Vibe Coding XR: Accelerating AI + XR Prototyping with XR Blocks and Gemini","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T00:17:41.692833Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2603.24591"},"observation_digest":"sha256:8dd6c01b1bcfd924866131a3866e4c8d237ee109e9e9f863d7d49e79956566e9","observation_id":"988ad47f-da48-4ce5-a6b1-6a13ba1ffa12","resolution":{"observed_at":"2026-05-15T00:18:21.868181Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.02371","last_updated":"2026-06-29T16:52:45Z","snapshot_observed_at":"2026-07-13T15:50:48.216413Z","submitted_at":"2026-03-31T04:41:01Z","title":"Internalized Reasoning for Long-Context Visual Document Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T23:53:19.148407Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.02371"},"observation_digest":"sha256:c15dc240c84f1da5164ae0b3b606cad8ae9c018664d2df78bae62888a67d13c0","observation_id":"23dbe2e5-3896-41a3-aa64-3dffed36e444","resolution":{"observed_at":"2026-05-13T23:53:28.175114Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-07-13T15:50:49.083652Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.02371","last_updated":"2026-06-29T16:52:45Z","snapshot_observed_at":"2026-07-13T15:50:48.216413Z","submitted_at":"2026-03-31T04:41:01Z","title":"Internalized Reasoning for Long-Context Visual Document Understanding","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-13T15:50:49.083652Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.02371"},"observation_digest":"sha256:097e86e8ded0b66bcd715e848951469231744d82b8cfb00080d5aaa0cd93924a","observation_id":"5696c388-44b9-4c34-95b3-4aa9c37d11a1","resolution":{"observed_at":"2026-07-13T15:50:49.083652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.04812","last_updated":"2026-04-06T16:16:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-06T16:16:24Z","title":"SysTradeBench: An Iterative Build-Test-Patch Benchmark for Strategy-to-Code Trading Systems with Drift-Aware Diagnostics","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T19:32:25.959711Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.04812"},"observation_digest":"sha256:7e2740c785d9524769f7f7022ac49d7a7d7047a223fab488394dce325ad25aef","observation_id":"9159a9c8-47fd-4f0b-91d5-d525b6c81d28","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.05426","last_updated":"2026-04-10T07:32:38Z","snapshot_observed_at":"2026-07-31T08:48:26.186922Z","submitted_at":"2026-04-07T04:40:17Z","title":"ALTO: Adaptive LoRA Tuning and Orchestration for Heterogeneous LoRA Training Workloads","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T20:21:27.428360Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.05426"},"observation_digest":"sha256:b8dc54a6336f9b3726d12fed55faf4fe19c1f04c77ed617cf08ca8a32b4ea1f4","observation_id":"c4e90c56-fcb3-4843-80dd-63f663e666f9","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.05460","last_updated":"2026-04-07T05:44:08Z","snapshot_observed_at":"2026-07-06T22:54:12.531121Z","submitted_at":"2026-04-07T05:44:08Z","title":"LLM Evaluation as Tensor Completion: Low Rank Structure and Semiparametric Efficiency","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-10T19:39:49.792609Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.05460"},"observation_digest":"sha256:fc667a0f81444ba8150c27a4bc058bfbc6a7276e5850c69135509fb791fad3ab","observation_id":"be803fee-5694-4c59-ad52-6e571d2bfe0e","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.05912","last_updated":"2026-04-07T14:15:45Z","snapshot_observed_at":"2026-07-06T22:54:30.567988Z","submitted_at":"2026-04-07T14:15:45Z","title":"FrontierFinance: A Long-Horizon Computer-Use Benchmark of Real-World Financial Tasks","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-10T19:49:32.983778Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.05912"},"observation_digest":"sha256:80b7f54e7df4fd109076a3de2fed09258132c94a64586a541cd2b9723a9c836c","observation_id":"20bd82eb-43f9-4303-a8c1-eb40cf2202ab","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.08588","last_updated":"2026-03-31T19:29:17Z","snapshot_observed_at":"2026-08-03T07:48:54.296680Z","submitted_at":"2026-03-31T19:29:17Z","title":"Act or Escalate? Evaluating Escalation Behavior in Automation with Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.08588"},"observation_digest":"sha256:99775621b38c8edf3269bf0446f88258e905d88d42d944350109149d86302ce3","observation_id":"9492a5d2-a2f2-4c93-91b7-894c9fe32c06","resolution":{"observed_at":"2026-05-13T23:28:25.839101Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.09444","last_updated":"2026-04-10T16:01:13Z","snapshot_observed_at":"2026-07-06T22:58:17.167367Z","submitted_at":"2026-04-10T16:01:13Z","title":"Confidence Without Competence in AI-Assisted Knowledge Work","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T17:02:54.050021Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.09444"},"observation_digest":"sha256:fb4e04f73059b7333d13289ddb1f2dcffa4c0a94968e25dbecc0f936458adac0","observation_id":"71ceab0a-c40c-4aba-b195-8f6c45c1ee4f","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.19394","last_updated":"2026-04-21T12:19:48Z","snapshot_observed_at":"2026-07-06T23:06:02.635464Z","submitted_at":"2026-04-21T12:19:48Z","title":"Can Continual Pre-training Bridge the Performance Gap between General-purpose and Specialized Language Models in the Medical Domain?","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-10T02:28:50.416827Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.19394"},"observation_digest":"sha256:902d7f77b680375554b9b9e66f73c4be7cb2a6c7b71a0b5685190cb0f915e25c","observation_id":"4e07459e-a70e-4115-a1d2-e45dd3cec6f4","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.23178","last_updated":"2026-06-24T13:27:28Z","snapshot_observed_at":"2026-08-03T03:16:46.165201Z","submitted_at":"2026-04-25T07:18:30Z","title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T08:14:18.535385Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.23178"},"observation_digest":"sha256:2e48dbe2a389da49a5b0dda05a1f773f06ea57690aac26bde8f49c05e977fd0b","observation_id":"2e5f4802-f1cd-4094-a3f2-f662c884fa60","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2604.26235","last_updated":"2026-04-29T02:32:14Z","snapshot_observed_at":"2026-08-02T11:16:41.718755Z","submitted_at":"2026-04-29T02:32:14Z","title":"LATTICE: Evaluating Decision Support Utility of Crypto Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-07T13:30:46.523784Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2604.26235"},"observation_digest":"sha256:735a2b4bc50f00d47a0492063bf3223f1266261f98603b14cc034d38970eefcd","observation_id":"42f5b3fa-cd29-4735-95a6-bbb94f22fbe9","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.02463","last_updated":"2026-05-06T08:34:04Z","snapshot_observed_at":"2026-07-06T23:15:32.349713Z","submitted_at":"2026-05-04T11:06:19Z","title":"When Stress Becomes Signal: Detecting Antifragility-Compatible Regimes in Multi-Agent LLM Systems","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-08T02:13:17.642291Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.02463"},"observation_digest":"sha256:172bac7edbd424ad4b580275f74793b8fe67e6c064a80cdd883043cfa916952b","observation_id":"3f95d9cb-6daf-4e65-98fa-1eb09ce21f6c","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.02463","last_updated":"2026-05-06T08:34:04Z","snapshot_observed_at":"2026-07-06T23:15:32.349713Z","submitted_at":"2026-05-04T11:06:19Z","title":"When Stress Becomes Signal: Detecting Antifragility-Compatible Regimes in Multi-Agent LLM Systems","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-08T18:29:00.554110Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.02463"},"observation_digest":"sha256:6c0a5d6d9c5a34a6a08f81a478425ab4ec98e1c812068fb40f76d1ff706ba415","observation_id":"4717e454-09b4-44c0-ab02-08fc9d983156","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.02930","last_updated":"2026-04-27T18:07:58Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-27T18:07:58Z","title":"Analysis and Explainability of LLMs Via Evolutionary Methods","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-09T20:37:40.932811Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.02930"},"observation_digest":"sha256:0a8b1975f85b0235a95d45c6770c1141f94f3b4c2d15ba098e2127f78ddf7eb9","observation_id":"c4ad8f3b-56fa-414b-b30c-5fcfc3798edd","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.03179","last_updated":"2026-05-04T21:42:10Z","snapshot_observed_at":"2026-08-03T23:09:27.517424Z","submitted_at":"2026-05-04T21:42:10Z","title":"A Validated Prompt Bank for Malicious Code Generation: Separating Executable Weapons from Security Knowledge in 1,554 Consensus-Labeled Prompts","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-08T18:11:29.066362Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.03179"},"observation_digest":"sha256:15c541f325d7dfa852f8160aefeded23a46c94bcd8bb24a3f8d23381ffc5e386","observation_id":"77f57b36-a566-443b-b2e3-887b9e40e599","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.04312","last_updated":"2026-05-05T21:24:58Z","snapshot_observed_at":"2026-07-06T23:17:04.200299Z","submitted_at":"2026-05-05T21:24:58Z","title":"Agent Island: A Saturation- and Contamination-Resistant Benchmark from Multiagent Games","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-08T17:06:32.814188Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.04312"},"observation_digest":"sha256:12df04ea5abd1305fd708c370c54885cb49230fed12a5b3cddecb676a9518379","observation_id":"bb8e06a7-9b25-46f0-b8aa-ef96652c4b41","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.05427","last_updated":"2026-05-30T15:07:19Z","snapshot_observed_at":"2026-07-06T23:18:02.987518Z","submitted_at":"2026-05-06T20:35:57Z","title":"The Refusal--Compliance Tradeoff: A Large-Scale Safety Behavior Audit of Large Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-08T16:44:31.583936Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.05427"},"observation_digest":"sha256:6b43b8ae4b6f0855c6326ead6ad6003da2a7d77ea734dc4d56bbf52542154ce5","observation_id":"0cfff2cd-51ae-4974-9bcd-58b9d64ccb60","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.08647","last_updated":"2026-05-09T03:35:09Z","snapshot_observed_at":"2026-07-06T23:20:52.521341Z","submitted_at":"2026-05-09T03:35:09Z","title":"AgentCollabBench: Diagnosing When Good Agents Make Bad Collaborators","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-12T00:54:36.868103Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.08647"},"observation_digest":"sha256:5af72e72309dd012418b8319228c0df6ec6408e40d1767cc205245c418c8bfa4","observation_id":"4ff60c3b-d11a-49a5-88fe-faf44a00eacc","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.09228","last_updated":"2026-05-09T23:56:04Z","snapshot_observed_at":"2026-07-06T23:21:21.560069Z","submitted_at":"2026-05-09T23:56:04Z","title":"ProactBench: Beyond What The User Asked For","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-12T02:14:01.145443Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.09228"},"observation_digest":"sha256:700617fcf184092efb943c6e3b1342fac68b05aa35db92ba9ee7baed801f989e","observation_id":"76c3cf12-7ab9-4eb5-a36f-b95dd1038667","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.09808","last_updated":"2026-05-10T23:06:24Z","snapshot_observed_at":"2026-08-02T12:27:41.948284Z","submitted_at":"2026-05-10T23:06:24Z","title":"Quantifying the Utility of User Simulators for Building Collaborative LLM Assistants","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T02:28:13.317630Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.09808"},"observation_digest":"sha256:53ad324f7085cd63991ad7db7d3fe2db2fa993b05a172265420b6cd1545fd5e2","observation_id":"4e8ea536-a7ad-4355-99ab-6f25de15f9f5","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-02T13:54:55.570407Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:79ce5777a737e323886bcbc3575a092023f54c7bca6cab17bf2214aeb7051202","observation_id":"c8001ea3-a8bb-45a2-9ad0-bbc0f631dabf","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.14679","last_updated":"2026-05-14T10:48:48Z","snapshot_observed_at":"2026-08-05T06:43:15.187811Z","submitted_at":"2026-05-14T10:48:48Z","title":"AI-assisted cultural heritage dissemination: Comparing NMT and glossary-augmented LLM translation in rock art documents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T21:07:08.026581Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.14679"},"observation_digest":"sha256:19a119ac0acc08e8059d7b3a2961e5b5e7ce6945f2bbee69b04c2761281b2cdf","observation_id":"6a208242-f5c2-488a-aa63-ec42280d6d09","resolution":{"observed_at":"2026-06-30T21:15:04.643040Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.15177","last_updated":"2026-05-17T19:31:24Z","snapshot_observed_at":"2026-07-06T23:26:32.979566Z","submitted_at":"2026-05-14T17:57:40Z","title":"OpenDeepThink: Parallel Reasoning via Bradley-Terry Aggregation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T03:06:54.507594Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.15177"},"observation_digest":"sha256:4944b9e89bb5a1407e9a7df262853b254ac9708a907f04305e4b1fa63bd2f57c","observation_id":"49371657-65f0-4fad-9a63-384681b87a43","resolution":{"observed_at":"2026-05-15T03:08:58.379385Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.15177","last_updated":"2026-05-17T19:31:24Z","snapshot_observed_at":"2026-07-06T23:26:32.979566Z","submitted_at":"2026-05-14T17:57:40Z","title":"OpenDeepThink: Parallel Reasoning via Bradley-Terry Aggregation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-20T20:51:47.393589Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.15177"},"observation_digest":"sha256:c1110aec87f04ba6d0014fc5c013f88149ed1728b1000a6b0e1af89e752715bc","observation_id":"dfbc38e3-72e3-4714-8fc4-92c05da62fdf","resolution":{"observed_at":"2026-05-20T20:53:43.622699Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.16381","last_updated":"2026-05-11T05:01:15Z","snapshot_observed_at":"2026-08-02T15:40:21.659895Z","submitted_at":"2026-05-11T05:01:15Z","title":"StreamPro: From Reactive Perception to Proactive Decision-Making in Streaming Video","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-20T23:32:38.332828Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.16381"},"observation_digest":"sha256:76c3e88af0012aabf3ca9f9b6f1abc94660614a5e50eaa44882e04bc38bdd498","observation_id":"0fba3910-90e3-47e3-8633-da161ca5bf27","resolution":{"observed_at":"2026-05-20T23:33:50.709916Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.17829","last_updated":"2026-05-18T04:03:18Z","snapshot_observed_at":"2026-08-05T06:31:55.615676Z","submitted_at":"2026-05-18T04:03:18Z","title":"Interactive Evaluation Requires a Design Science","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T10:55:08.135630Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.17829"},"observation_digest":"sha256:2e9f2113d7123ee66bfdcb46f59f8cdaa0c6b5ecb283010c6797e7fc9b622910","observation_id":"302d5554-ffb1-41ba-b394-9d0bd85461d8","resolution":{"observed_at":"2026-05-20T10:58:14.097172Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.18357","last_updated":"2026-05-18T13:12:30Z","snapshot_observed_at":"2026-08-01T09:43:26.451580Z","submitted_at":"2026-05-18T13:12:30Z","title":"Engagement vs. Commitment: The Economic Trade-Offs of Polarizing News Content","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-19T23:30:19.352088Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.18357"},"observation_digest":"sha256:790406bf9211d3d49ac622a983e8afd772c8b40911f0c863eb9f7b1a017d2a06","observation_id":"58d3fcb2-1f2b-421e-a77f-3b0e5ac76424","resolution":{"observed_at":"2026-05-19T23:32:52.791540Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.18630","last_updated":"2026-05-18T16:34:45Z","snapshot_observed_at":"2026-07-06T23:29:29.105126Z","submitted_at":"2026-05-18T16:34:45Z","title":"SCICONVBENCH: Benchmarking LLMs on Multi-Turn Clarification for Task Formulation in Computational Science","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-20T10:36:09.724234Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.18630"},"observation_digest":"sha256:a6e799374fff00189cc362da6619c2b23388d7287c01b262ce0eae7e03350937","observation_id":"03363f7a-2574-4f4d-8b89-4f6839531ace","resolution":{"observed_at":"2026-05-20T10:38:12.069195Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.23262","last_updated":"2026-05-22T06:03:01Z","snapshot_observed_at":"2026-08-02T00:22:40.506306Z","submitted_at":"2026-05-22T06:03:01Z","title":"Design and Report Benchmarks for Knowledge Work","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-25T04:39:14.319133Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.23262"},"observation_digest":"sha256:8a5a1c2f972dcd1247c75245d0258b630f35bdc1303a618691c22f3968f111f6","observation_id":"cd68035e-29db-4470-a76f-de948ef291a3","resolution":{"observed_at":"2026-05-25T04:40:23.466577Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.24279","last_updated":"2026-05-22T23:13:21Z","snapshot_observed_at":"2026-07-31T10:30:03.841807Z","submitted_at":"2026-05-22T23:13:21Z","title":"ContextEcho: A Benchmark for Persona Drift in Long Agentic-Coding Sessions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-30T15:17:37.904831Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.24279"},"observation_digest":"sha256:40132777f88438bc1ba7aa82cef43a593892e2e9501890823e254b9301b28e43","observation_id":"cb50ee5f-f3f8-446a-bba2-571b487fadd8","resolution":{"observed_at":"2026-06-30T15:34:48.434672Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.25272","last_updated":"2026-05-24T21:59:08Z","snapshot_observed_at":"2026-07-31T09:15:58.688614Z","submitted_at":"2026-05-24T21:59:08Z","title":"AI Cartography: Mapping the Latent Landscape of AI Benchmark Ecosystems","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.25272"},"observation_digest":"sha256:80904cc54a255670a6c2bbdcaf5be0e70a07db399ca0cf72c701650b8ccf179d","observation_id":"5dddc29f-a976-4ff4-a8ae-fb3b6f52ebb3","resolution":{"observed_at":"2026-06-30T00:24:04.104979Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.27914","last_updated":"2026-06-09T02:24:01Z","snapshot_observed_at":"2026-07-06T23:37:36.244456Z","submitted_at":"2026-05-27T03:41:11Z","title":"Does Capability Transfer to Subjective Behavior -- and Would Our Instruments Tell Us? A Self-Evolving, Trust-by-Construction Evaluation Paradigm","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T12:54:36.818698Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.27914"},"observation_digest":"sha256:107a19cbfa27e723b0c0f1bd78c058428119be4457ddcf8d66e9e6aab2d1312e","observation_id":"e1651dd1-5a50-4a9a-aa2a-022c2cd756a8","resolution":{"observed_at":"2026-06-29T13:03:26.793885Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.29395","last_updated":"2026-05-28T05:44:43Z","snapshot_observed_at":"2026-08-03T18:23:05.064909Z","submitted_at":"2026-05-28T05:44:43Z","title":"Low Rank for Rank: Uncertainty-Aware Task-Specific LLM Ranking under Sparse Pairwise Comparisons","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T06:11:48.446361Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.29395"},"observation_digest":"sha256:b7e4a04a0afc8a3731dce6703ef3b53cf34edad41515df681299c132539831a1","observation_id":"371314e1-cd82-4728-9da0-8b5567f94400","resolution":{"observed_at":"2026-06-29T14:53:32.261037Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.29473","last_updated":"2026-06-21T04:01:36Z","snapshot_observed_at":"2026-07-06T23:38:56.014434Z","submitted_at":"2026-05-28T07:04:56Z","title":"Inform, Coach, Relate, Listen: Auditing LLM Caregiving Support Roles","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-29T05:54:39.899517Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.29473"},"observation_digest":"sha256:6d2b3715858cf94f380d0c1698ce1f56a7244605b1b3f06a15bfd560ff0b3b5a","observation_id":"69bbf51c-54e3-4b1d-a7e6-31ef517638ff","resolution":{"observed_at":"2026-06-29T06:03:08.867810Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.30758","last_updated":"2026-05-29T02:41:18Z","snapshot_observed_at":"2026-08-03T06:24:52.704941Z","submitted_at":"2026-05-29T02:41:18Z","title":"Pairwise Reference Alignment as a Model-Level Ordinal Observable","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T22:46:11.450812Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.30758"},"observation_digest":"sha256:fa452faf24fa88153d37b3fa427484b610672a0cdf25320e174eadc6ed7f3053","observation_id":"4e2e5a9c-e207-4e35-8b97-0e67991d320c","resolution":{"observed_at":"2026-07-01T19:25:59.686950Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2605.31080","last_updated":"2026-05-29T09:47:45Z","snapshot_observed_at":"2026-07-06T23:40:17.605034Z","submitted_at":"2026-05-29T09:47:45Z","title":"A Pilot Study on Curator-Guided Multilingual Art Description for Blind and Low-Vision Audiences with Small Vision-Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T19:53:36.990878Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2605.31080"},"observation_digest":"sha256:424491ef4ee625f715ffdc62d6d29445b29c6736bfe2af58a713184fd8bf53fe","observation_id":"5f2425e7-1786-48d5-96c4-41f0fba894f9","resolution":{"observed_at":"2026-06-28T20:22:37.608482Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.00931","last_updated":"2026-05-30T23:37:55Z","snapshot_observed_at":"2026-08-01T00:57:33.201756Z","submitted_at":"2026-05-30T23:37:55Z","title":"CV-Arena: An Open Benchmark for Instructional Computer Vision Problem Solving with Human-AI Collaborative Preferences","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T18:36:46.381564Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.00931"},"observation_digest":"sha256:a4162af331983ac0725fde27a2dbe166273b39304f2400da8f9f854efc67df37","observation_id":"cff335cc-ace1-47a5-929b-402bda8803e9","resolution":{"observed_at":"2026-06-28T20:32:37.621567Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.01034","last_updated":"2026-05-31T05:50:27Z","snapshot_observed_at":"2026-08-04T18:54:04.912017Z","submitted_at":"2026-05-31T05:50:27Z","title":"A Finite-Calibration Regime Map for LLM Judge Panels","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T17:34:28.258222Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.01034"},"observation_digest":"sha256:616598969785347b1908aed209c9c1e06a96635205526537a9f6101c74ca0aa4","observation_id":"adc9565e-4185-48aa-bb93-651eb5b1c336","resolution":{"observed_at":"2026-07-01T21:06:13.143303Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.01136","last_updated":"2026-05-31T10:15:36Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T10:15:36Z","title":"From Outliers to Errors: Auditing Pali-to-English LLM Translations with Multi-Reference Adjudication","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T17:11:12.419269Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.01136"},"observation_digest":"sha256:838d1e73b9a1b6e585edd84ef75819539abb471329b0b3f131b3aaf5d28e63ce","observation_id":"d58710c6-de9a-497d-b2c0-3b34d70bcbed","resolution":{"observed_at":"2026-06-28T17:12:24.309494Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:89bd388a5b26c2d0486d64535d83cd8b9a4e64532f3e2f1d5955af64de938c7f","observation_id":"915bf9c3-9d18-4d79-9a82-343735faa7cf","resolution":{"observed_at":"2026-07-02T03:36:29.207143Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.04273","last_updated":"2026-06-02T22:58:19Z","snapshot_observed_at":"2026-07-06T23:44:23.633099Z","submitted_at":"2026-06-02T22:58:19Z","title":"Characterizing initial human-AI proof formalization workflows","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-28T09:29:50.282874Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.04273"},"observation_digest":"sha256:7ea2cc1047ac3266eac182f32b07d2dee2a643bc5658f1384f7a222bbc40b6bc","observation_id":"29a0f2b7-83f8-4a06-bbe1-5e1890a7a32d","resolution":{"observed_at":"2026-07-02T04:06:34.849422Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.05384","last_updated":"2026-06-03T19:37:23Z","snapshot_observed_at":"2026-08-03T19:30:59.858009Z","submitted_at":"2026-06-03T19:37:23Z","title":"Stability vs. Manipulability: Evaluating Robustness Under Post-Decision Interaction in LLM Judges","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-06-28T05:58:59.870335Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.05384"},"observation_digest":"sha256:588d497fcc88ef99c446c96f091d21ce6cde059216b29a90228f14bb6bf8de49","observation_id":"65b99aeb-3ec9-4ba0-99a1-81f7c3d6c7bb","resolution":{"observed_at":"2026-07-02T08:36:47.673444Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.06546","last_updated":"2026-06-04T07:40:12Z","snapshot_observed_at":"2026-07-31T21:40:27.810095Z","submitted_at":"2026-06-04T07:40:12Z","title":"Elmes*: Automated Construction of Fine-Grained Evaluation Rubrics for Large Language Models in Long-Tail Educational Scenarios","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-28T02:49:01.527149Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.06546"},"observation_digest":"sha256:a472d2e912f6f88f564615847632ea2095a2432188dd2f1fa42746979831a299","observation_id":"ed6470c8-fb15-4353-b068-1ea25f26b8ab","resolution":{"observed_at":"2026-07-02T11:56:55.169551Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.07492","last_updated":"2026-06-05T17:46:36Z","snapshot_observed_at":"2026-08-03T21:10:23.656676Z","submitted_at":"2026-06-05T17:46:36Z","title":"Bradley-Terry Rankings for Recommender Systems Across Dataset Taxonomies","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T20:25:42.500234Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.07492"},"observation_digest":"sha256:8635e81a52c089b324fef58ebeaa85e7fe283dd4ffe84daa33d4c5020157449c","observation_id":"d5233192-3592-4e6b-b655-325d90713e30","resolution":{"observed_at":"2026-07-02T20:27:22.393053Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.10064","last_updated":"2026-06-08T18:39:15Z","snapshot_observed_at":"2026-08-02T04:32:36.214852Z","submitted_at":"2026-06-08T18:39:15Z","title":"Bittensor Agent Arenas as a Trajectory Primitive: Distilling a Shopping Agent from ShoppingBench Subnet Traces","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T17:07:39.645843Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.10064"},"observation_digest":"sha256:902b9a6432e91f479ab6022e11f1906cf7c4394693734b9abe57f1273d5997e9","observation_id":"b6540e42-6e1c-41a7-a179-3745e148d073","resolution":{"observed_at":"2026-07-03T00:37:29.845249Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.12702","last_updated":"2026-06-10T21:44:20Z","snapshot_observed_at":"2026-07-31T15:03:20.136750Z","submitted_at":"2026-06-10T21:44:20Z","title":"Deployment-Centered Evaluation: Predicting Query-Level Rejection Risk in a Clinical LLM System","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T09:43:12.633039Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.12702"},"observation_digest":"sha256:59a9c85aa93f3ed0a0c25d54c58b577bb270c9945af7b8ac77249c09ed16146a","observation_id":"dbcd1090-bd88-4a5e-82a7-4448e0ef5fd0","resolution":{"observed_at":"2026-07-03T11:08:02.942190Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.19714","last_updated":"2026-06-18T02:26:05Z","snapshot_observed_at":"2026-08-01T18:27:22.876701Z","submitted_at":"2026-06-18T02:26:05Z","title":"AURA: Adaptive Uncertainty-aware Refinement for LLM-as-a-Judge Auditing","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-26T15:48:26.303462Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.19714"},"observation_digest":"sha256:0e5e229d5dc67753fb224462f10c301abdb94b14446047c7af748729a000387d","observation_id":"01eb3df8-1d6f-472e-b7a4-229b3150dc6a","resolution":{"observed_at":"2026-07-04T05:39:40.119444Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.20241","last_updated":"2026-06-18T13:52:37Z","snapshot_observed_at":"2026-08-04T03:01:55.342021Z","submitted_at":"2026-06-18T13:52:37Z","title":"BAFIS: Dataset + Framework to assess occupational Bias and Human Preference in modern Text-to-image Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T17:53:46.206401Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.20241"},"observation_digest":"sha256:1d6347fe66eb67726ccfa2fee65dbc154978a8a523aa0813436bd0eef65d4b35","observation_id":"26ca8f70-4bbd-4c25-9af6-baf3bdc742f5","resolution":{"observed_at":"2026-07-04T03:39:29.446244Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.21460","last_updated":"2026-06-19T14:16:13Z","snapshot_observed_at":"2026-07-06T23:56:26.805329Z","submitted_at":"2026-06-19T14:16:13Z","title":"Evaluation of Small Language Models for Arabic Language Processing","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-26T14:22:37.152936Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.21460"},"observation_digest":"sha256:2ee3eb354a346cdb7e8b496c8b322cfc0d88ff51867e090b406dc793264bc4b7","observation_id":"8a3c0d88-b33e-4a02-b932-f269fcf3946e","resolution":{"observed_at":"2026-07-04T06:39:36.969233Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.26348","last_updated":"2026-06-24T19:40:53Z","snapshot_observed_at":"2026-07-07T00:00:41.502665Z","submitted_at":"2026-06-24T19:40:53Z","title":"What We are Missing in Multimodal LLM Evaluation?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-26T01:31:35.307417Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.26348"},"observation_digest":"sha256:329bd4000b011b6323758c3cd26945256efe0e258c06b36a086c604e01a4ada9","observation_id":"fac787c1-5121-4ab7-aa74-f87fad71e94e","resolution":{"observed_at":"2026-07-04T15:39:56.690059Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.28365","last_updated":"2026-06-14T16:59:18Z","snapshot_observed_at":"2026-07-07T00:02:29.000757Z","submitted_at":"2026-06-14T16:59:18Z","title":"CAMI: Cost-Aware Agent-Guided Multi-Indexing for Semantic Retrieval","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-30T11:07:31.861143Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.28365"},"observation_digest":"sha256:1092626c78f3b67731fbe5c8f55a0fb0d8b3cf91a5d4cf4764018fb6b51d474d","observation_id":"5d2ea79c-9aff-4a06-bbeb-4fca48a15fcd","resolution":{"observed_at":"2026-06-30T11:24:38.797790Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.28960","last_updated":"2026-06-27T15:00:20Z","snapshot_observed_at":"2026-08-01T15:07:26.881714Z","submitted_at":"2026-06-27T15:00:20Z","title":"Expert Evaluation of Clinical AI Tools on Real Point-of-Care Clinical Queries","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-30T09:30:00.467709Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.28960"},"observation_digest":"sha256:47af7ae35c5198a03b4a04492817a5d3985d1db95caea03bc343806fa1b072e3","observation_id":"c91fc324-c48b-4935-b971-07f1eb83269b","resolution":{"observed_at":"2026-06-30T09:34:34.718843Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.29784","last_updated":"2026-06-29T05:04:57Z","snapshot_observed_at":"2026-08-01T20:11:56.075450Z","submitted_at":"2026-06-29T05:04:57Z","title":"HERO: Improving the Reliability and Sensitivity of Generative Model Evaluation Using Historical Data","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T05:41:56.435040Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.29784"},"observation_digest":"sha256:8268b89c90bea3ce4cdc6595b1c7b196eb893435d2fb7f8489f4d6e579326ff6","observation_id":"667234eb-960d-40bd-80e8-99a2519905b3","resolution":{"observed_at":"2026-06-30T13:54:44.499146Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.30116","last_updated":"2026-06-29T10:47:28Z","snapshot_observed_at":"2026-07-07T00:04:00.652252Z","submitted_at":"2026-06-29T10:47:28Z","title":"Open Problems in Constitutional Preference Reconstruction","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T06:49:24.255622Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.30116"},"observation_digest":"sha256:f63048baf9ba9248366954b46df0331073846827201470ef2355f94ffee44f7b","observation_id":"e0ec5ac9-52c8-405b-83b4-d5bdbd98e582","resolution":{"observed_at":"2026-06-30T06:54:20.510382Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.30336","last_updated":"2026-06-30T07:03:28Z","snapshot_observed_at":"2026-08-03T04:09:35.000794Z","submitted_at":"2026-06-29T14:14:34Z","title":"FlexTab: A Flexible Encoder-Decoder Architecture for In-Context Learning Across Diverse Tabular Tasks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-30T07:22:32.638556Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.30336"},"observation_digest":"sha256:0464ae92dfd94aacfe03644e7e8a7f42d0b3dec7a0a6123e3199ffa6d6edfad9","observation_id":"9886c68a-673e-4848-8c85-c76604d6b443","resolution":{"observed_at":"2026-06-30T07:24:21.045668Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.30336","last_updated":"2026-06-30T07:03:28Z","snapshot_observed_at":"2026-08-03T04:09:35.000794Z","submitted_at":"2026-06-29T14:14:34Z","title":"FlexTab: A Flexible Encoder-Decoder Architecture for In-Context Learning Across Diverse Tabular Tasks","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-01T07:13:43.014623Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.30336"},"observation_digest":"sha256:ce7de333ea4277d3721e5ceed524e59418a5725d4102b16155a086a2d9ee4a1f","observation_id":"4c8a0f39-84f9-4b74-b096-afe663c86cb6","resolution":{"observed_at":"2026-07-01T07:15:29.983182Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2607.01740","last_updated":"2026-07-02T05:52:32Z","snapshot_observed_at":"2026-07-07T00:07:19.026006Z","submitted_at":"2026-07-02T05:52:32Z","title":"Meta-Benchmarks for Financial-Services LLM Evaluation","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-07-03T14:08:20.432931Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.01740"},"observation_digest":"sha256:f1a78b0b2d79ce99405b5c377792b662b72432d979c6e81fa144829d5002cb3c","observation_id":"1b132ae3-9955-43be-9d57-3142f906bbcd","resolution":{"observed_at":"2026-07-03T14:18:22.689464Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2607.07003","last_updated":"2026-07-24T00:45:33Z","snapshot_observed_at":"2026-08-03T11:48:08.490751Z","submitted_at":"2026-07-08T04:54:54Z","title":"Dissociating the Internal Representations of Sycophancy in LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-09T21:56:32.778349Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.07003"},"observation_digest":"sha256:b4663df53664c2b370e20638983a3b16bd6c07df4c9eb587a678467db9af886a","observation_id":"f8c96aa6-ab97-44c2-9693-257764ebc289","resolution":{"observed_at":"2026-07-09T22:06:35.339752Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-02T08:11:39.471149Z","title":"N., Li, T., Li, D., Zhang, H., Zhu, B., Jordan, M., Gonza- lez, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.07003","last_updated":"2026-07-24T00:45:33Z","snapshot_observed_at":"2026-08-03T11:48:08.490751Z","submitted_at":"2026-07-08T04:54:54Z","title":"Dissociating the Internal Representations of Sycophancy in LLMs","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T08:11:39.471149Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.07003"},"observation_digest":"sha256:0178fc330e4614984842bdac39817573b666dccc71bb5240db80d3210754a41e","observation_id":"f7c91878-0858-4415-8d2e-646c3601dea8","resolution":{"observed_at":"2026-08-02T08:11:39.471149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2607.08758","last_updated":"2026-07-09T17:55:53Z","snapshot_observed_at":"2026-07-12T23:18:59.558861Z","submitted_at":"2026-07-09T17:55:53Z","title":"Ideas Have Genomes: Benchmarking Scientific Lineage Reasoning and Lineage-Grounded Idea Generation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-10T01:51:14.922841Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.08758"},"observation_digest":"sha256:9611235db166c6622cd88c2e6da95f72a5499e4193aef3adaacdbe2f5fd4d35a","observation_id":"2f6f0dd8-f2db-48bd-b322-a6e7c8081cec","resolution":{"observed_at":"2026-07-10T01:56:41.111012Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-02T06:14:06.034941Z","title":"Chatbot arena: An open platform for evaluating llms based on human preference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13125","last_updated":"2026-07-18T17:28:17Z","snapshot_observed_at":"2026-08-03T17:18:33.842709Z","submitted_at":"2026-07-14T17:52:05Z","title":"Boogu-Image-0.1: Boosting Open Agentic Multimodal Generation via Understanding under a Minimal Budget","version":2},"reference_index":145,"source":"pdf_text","source_observed_at":"2026-08-02T06:14:06.034941Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.13125"},"observation_digest":"sha256:7d7ce1f33fd795ed524cfeded800f66ca3e801b0cb9f3e9e8729adaf6e331cc9","observation_id":"0eb6e539-f6d7-4683-a86c-f3d8776d222e","resolution":{"observed_at":"2026-08-02T06:14:06.034941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-02T05:59:51.029265Z","title":"Jordan, Joseph E","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13189","last_updated":"2026-07-14T18:39:33Z","snapshot_observed_at":"2026-08-02T05:59:49.522074Z","submitted_at":"2026-07-14T18:39:33Z","title":"RAGthoven at SemEval-2026 Task 1: A Multi-Stage Pipeline Walks Into a Benchmark and Barely Clears the Bar","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-02T05:59:51.029265Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.13189"},"observation_digest":"sha256:cb314407d5370b7f06c368a839b53ff57276abf2d19bb20fc8f3f0c81b6d4c93","observation_id":"d3695e37-570c-45da-9ddc-0488959a07f7","resolution":{"observed_at":"2026-08-02T05:59:51.029265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-01T22:06:31.303455Z","title":"Chatbot arena: An open platform for evaluating llms by human preference.arXiv preprint arXiv:2403.04132, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15899","last_updated":"2026-07-17T12:12:38Z","snapshot_observed_at":"2026-08-02T11:39:51.456302Z","submitted_at":"2026-07-17T12:12:38Z","title":"ContinuityBench: A Benchmark and Systems Study of Stateful Failover in Multi-Provider LLM Routing","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T22:06:31.303455Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.15899"},"observation_digest":"sha256:1cde8243a31f6f3c7a6adebb94464d218f673bb3150648fe318c88c2faed620a","observation_id":"5dc02a58-a58b-4efd-80a9-a74db86a856e","resolution":{"observed_at":"2026-08-01T22:06:31.303455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-01T15:29:25.040725Z","title":"Chatbot Arena: An Open Platform for EvaluatingLLMsbyHumanPreference","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18438","last_updated":"2026-07-20T18:46:17Z","snapshot_observed_at":"2026-08-04T02:52:47.040983Z","submitted_at":"2026-07-20T18:46:17Z","title":"Relay-Bench: Evaluating LLMs on Multi-Domain Reasoning Chains","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T15:29:25.040725Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.18438"},"observation_digest":"sha256:6655aca8db68029b8a2205d476c8777b7e3bdf5ee7c31cae9a621a730a68a794","observation_id":"1afa2c31-87b0-4c15-858f-f118faad26df","resolution":{"observed_at":"2026-08-01T15:29:25.040725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-02T09:51:06.039948Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19375","last_updated":"2026-06-26T22:26:40Z","snapshot_observed_at":"2026-08-03T02:46:20.183203Z","submitted_at":"2026-06-26T22:26:40Z","title":"Economic Evaluations of Language Models","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-02T09:51:06.039948Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.19375"},"observation_digest":"sha256:1b23368a122c387c7b7e1ab169558c8676404b73ce5306507f935bc261616408","observation_id":"b7b015e0-5c66-4bcb-bb08-a301f121422c","resolution":{"observed_at":"2026-08-02T09:51:06.039948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-01T09:12:22.182908Z","title":"https://arxiv.org/abs/2403.04132v1","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.20862","last_updated":"2026-07-23T02:39:11Z","snapshot_observed_at":"2026-08-01T09:12:21.443648Z","submitted_at":"2026-07-23T02:39:11Z","title":"CSPF: A Constrained Shared-Private Fusion Method for Non-Verifiable Preference Evaluation","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-01T09:12:22.182908Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.20862"},"observation_digest":"sha256:6e69d7cecd8e89a58aadde5123e720f233aaa06e5439bac93798251d7fdaaee0","observation_id":"ecea5ad7-4b16-48b1-9baa-cd0ccf4f2c87","resolution":{"observed_at":"2026-08-01T09:12:22.182908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-01T05:03:44.347887Z","title":"arXiv preprint arXiv:2403.04132 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22368","last_updated":"2026-07-24T14:55:19Z","snapshot_observed_at":"2026-08-01T05:03:37.778828Z","submitted_at":"2026-07-24T14:55:19Z","title":"Do Agent Benchmarks Measure Capability? Protocol Validity in the Age of Agentic AI","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-01T05:03:44.347887Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.22368"},"observation_digest":"sha256:f96b6fbe8dae2773cf35e4afa32e730ce88d408933662b716301ed87f20f537e","observation_id":"03550489-49bf-4da5-b5e5-d9163a366e47","resolution":{"observed_at":"2026-08-01T05:03:44.347887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-01T07:40:15.001814Z","title":"and Gonzalez, Joseph E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27420","last_updated":"2026-07-29T19:42:26Z","snapshot_observed_at":"2026-08-04T10:55:46.147631Z","submitted_at":"2026-07-29T19:42:26Z","title":"Dimensionality and Measurement Precision in HLE's Multiple-Choice Subset","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-01T07:40:15.001814Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2607.27420"},"observation_digest":"sha256:abd4cf26f93dd206a48f118da4171e007f1b611b0ba29b43a184bf77f34b4fb7","observation_id":"48237d8a-376c-4304-87b9-3336c346a1e5","resolution":{"observed_at":"2026-08-01T07:40:15.001814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T04:18:53.743378Z","title":"Chiang, L","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00794","last_updated":"2026-08-04T16:49:21Z","snapshot_observed_at":"2026-08-05T07:53:59.747912Z","submitted_at":"2026-08-01T17:50:12Z","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T04:18:53.743378Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2608.00794"},"observation_digest":"sha256:90820f3784e0a0b955dbd36d714d3e1d4dd5706fcab7703dbe4ae4b16841a319","observation_id":"f7d1caaf-2e4b-46f2-b385-791fb0520842","resolution":{"observed_at":"2026-08-05T04:18:53.743378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.04132/citation-record","integrity":"/paper/2403.04132/integrity","json":"/paper/2403.04132/citation-record.json","paper":"/paper/2403.04132"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-04T15:46:25.710484Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:9bb70da533967d663c58e79512a70404ac26ab4f903b37531a60e4687dd51f86","observation_id":"3b852332-53e7-4b07-9e7f-5afdd21562e3","resolution":{"observed_at":"2026-05-13T15:13:25.347366Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"aos/1079120","doi":"10.1214/aos/1079120126","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Statistical behavior and consistency of classification methods based on convex risk minimization","venue":"The Annals of Statistics","work_id":"f624ca9b-b854-4cbe-98c1-6de111a6f482","year":2002},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:c18b2663ca0307361192f8475f76283828d82c1a4a1f4d57235b00ee5b651282","observation_id":"d142e643-0925-47b4-9744-8455131e05f0","resolution":{"observed_at":"2026-05-13T15:13:25.353915Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.findings-emnlp","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"emnlp-main.608","venue":null,"work_id":"479a7217-7b7d-4377-9e24-9b7ac69e31d5","year":2023},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:4739892c52a51deb0be6ddaa671704bf9c0feb9caa334cd5748150085a06a392","observation_id":"787e3bc8-c303-4384-aae9-92022ddcfd50","resolution":{"observed_at":"2026-05-13T15:13:25.357448Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-15T19:20:38.929511+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T19:20:38.929511+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:5939eed7ffac1a3187b52c4105a5419a040f636d38e51dd3cf71fff475519081","observation_id":"5a3589a1-94a7-4f14-a295-2b476f74e970","resolution":{"observed_at":"2026-05-13T15:13:25.362893Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":"2307.09288","doi":"10.24963/ijcai.2025/706","metadata_source":"pith","pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","venue":"cs.CL","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","year":2023},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:b63a080c0cdd988d85a4af842b1968ec60f0e3cf22b7cf6e01c2b65014fed89e","observation_id":"35aee26c-adcd-40c8-b846-637c71cf6786","resolution":{"observed_at":"2026-05-13T15:13:25.368965Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Travel Itinerary Planning","venue":null,"work_id":"a2d44760-aa80-44e1-8330-acdfb9da9fc4","year":1932},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:ea20a7be41d9364b11c143235da03e51ece6f5f901bdf132ee87c3121f1d737b","observation_id":"b8f8976f-bf49-416e-aa11-65afed9d7222","resolution":{"observed_at":"2026-05-13T15:13:25.372973Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"It houses collections of European paintings, a medieval and Renaissance collection, ceramics, French sculptures and more","venue":null,"work_id":"0626a3c6-a98e-4d59-a6fe-e82ff3c84f13","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:77c7ad5a634190e3de7a7211a1aa4b06fb0fade51a72af654564b22e344ae7c4","observation_id":"a3a773f7-2951-4bdb-b665-571133c1c05e","resolution":{"observed_at":"2026-05-13T15:13:25.377708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"edd120e3-41d1-4ebb-b8d5-aaa1f7d517e0","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:da4dbaf97003d7ca3118dbd70a1095387845424b2889958a727323b6d733027f","observation_id":"ff5e5e2e-da36-4b7d-9be5-474fc07dabaf","resolution":{"observed_at":"2026-05-13T15:13:25.381544Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"383e64e1-517d-4450-a3c7-661a8044c751","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:a93659536cd04410f5a88486ad084155e65a2c0575facf94c5a03b05e7330922","observation_id":"04aaa260-9393-47be-8f71-d584be635ecd","resolution":{"observed_at":"2026-05-13T15:13:25.384969Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5a179568-a3f7-48db-a2f2-b40d254249f0","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:dd11b44c4314bb82f0fe679e5674766b8199a54ebdb35da84c70d16b455722ce","observation_id":"7c6ceb9f-17b8-4e34-bc67-3cb6bddda03f","resolution":{"observed_at":"2026-05-13T15:13:25.389165Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"54c918a9-5c31-4492-916d-edec60f93b42","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:b10019874cfc905985327f1c90d49235bf8f9e8c2ad5a82907b95f726141d571","observation_id":"2956a5a4-2fa4-4f63-945f-3f35be0c6caa","resolution":{"observed_at":"2026-05-13T15:13:25.392667Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"016c5028-0661-4c07-8342-433b46d3db11","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:834b9be4350d72bced7d9857af4ba5c319ea7c3a4ff57e981abc0df8321018d7","observation_id":"3b60a4b1-ed88-4dd8-8c66-3867ec5d765c","resolution":{"observed_at":"2026-05-13T15:13:25.395924Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"64fba8b7-6170-4c81-a37c-f333065000bf","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:2f3c17183965f4e973cc7eecf4e91e426cc185327ce42a7ed5e5167bf53c4fe7","observation_id":"88f3a4e1-4bf7-4b77-bf07-7d0a3c52fa44","resolution":{"observed_at":"2026-05-13T15:13:25.399098Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"873749e9-0409-4727-99db-fecf479e0313","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:ba5840c27348a6ef183a933962b76249f4586cc9d9a707aab7bed45a35ff0c96","observation_id":"b9c0c2fa-ce7e-4ad9-a487-74682a1c16f9","resolution":{"observed_at":"2026-05-13T15:13:25.402166Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"eb4079bb-517a-4082-9c19-59204fbeac14","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:ae988732ea4aa6f2bb94541ea508a5c06a382a93d73c9c6da0b4a4625aac0bb1","observation_id":"59899fac-74b0-4cb2-952c-7f4fa5da14cb","resolution":{"observed_at":"2026-05-13T15:13:25.404987Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0149b3b4-22b0-4d48-b5f1-58567bc34b5e","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:3f70e937f651416bfba407b090c95ae1413118078b10274171f4c052f3405071","observation_id":"8ed6558b-229d-4c0d-bde5-41cceede1b47","resolution":{"observed_at":"2026-05-13T15:13:25.408873Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3dfa91fd-53a2-43d4-834b-66ac4c1364d7","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:f20c3dde4f16e2553670bc5c13e85160a85227ffce6ddb61e7cde3fbb9d168cf","observation_id":"dce4dfaa-a604-42bc-bd6a-068216c35d74","resolution":{"observed_at":"2026-05-13T15:13:25.412629Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e81d454e-178e-4e16-9742-f21df7a9ec3c","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:e2b6a3e589223952423753e71d495163ae26cc52a767f28e659311a1f6996a35","observation_id":"2d465c1f-d45a-4de6-bc5a-c1c8ea01fcfc","resolution":{"observed_at":"2026-05-13T15:13:25.416107Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a9aa806f-0a8e-45af-a6f6-1a6331fa73e8","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:41770d7551fd78ea84745f1d29b244363672d904b5fbae01173e370c5d1ba7e3","observation_id":"3f2c0d46-da3a-4aaf-ac5f-1d054d136ad5","resolution":{"observed_at":"2026-05-13T15:13:25.419406Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"88cbe2a8-c989-4b72-bd7b-191da63eb842","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:e1c9694d6690a295296dd47bd049b458c39648308037ba4a45747e34edd3b30d","observation_id":"abc05d99-10c8-4086-acfd-e241ee92cd02","resolution":{"observed_at":"2026-05-13T15:13:25.422286Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Remember to check the opening times and any COVID-19 restrictions before you visit","venue":null,"work_id":"5cdb1544-946d-4ec6-8cba-7eca30492192","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:3697ce38f86cc1b460907b19e4e86149531dff7f8d29b49ae4467c3b05b1b5f5","observation_id":"a5e625f3-9fc7-447b-97cc-a324ac44f331","resolution":{"observed_at":"2026-05-13T15:13:25.425537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fde29526-b316-4e15-98a7-5481d0e4c0b8","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:246c041d18a9daae61960661b0b644949f0f63278b67b77e4e25c47cea6b625a","observation_id":"ef59e6a6-e601-4c3a-88d8-a99d3c1029c3","resolution":{"observed_at":"2026-05-13T15:13:25.428646Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7e10242c-5b30-4d52-8cee-e19ef6c6a559","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:0bb5b6f2e3744ddc83893b2c1c65a89bba523524eb2227b0754cf3f73bbf43eb","observation_id":"a11181a5-6c78-4a8a-8e76-b6af720e4c12","resolution":{"observed_at":"2026-05-13T15:13:25.431553Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0098496f-b9d2-4976-948a-876438e7bfa5","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:6ad1534c8dd661e491b9ece87123df7d442c1bf23744f5e17a7c65a88fd903a0","observation_id":"c7e98b0d-fc78-495b-92f9-61310a62788e","resolution":{"observed_at":"2026-05-13T15:13:25.434909Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ea11db1c-d92a-4a66-887b-b9de66dfbc27","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:a6016f3392c520f910b4313729d36b88024ebe0567c143278707885d5fc7d489","observation_id":"d8403948-795c-4bec-a05a-a42863595158","resolution":{"observed_at":"2026-05-13T15:13:25.438581Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d405fbf7-0466-4474-8e87-e57cadbf030f","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:0323a97f34c98c1f74036349628930582013f198536bc3abd467f17843435b77","observation_id":"19aa2493-1f99-4adb-8086-7d40831e636f","resolution":{"observed_at":"2026-05-13T15:13:25.442218Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cc35b3ba-8c90-4add-a1c9-7aeea12125d5","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:f90a78b08adea4fe36299f69f17409fcfff391b7d7ee2148075b5917ab3ce1d6","observation_id":"bd4c204e-295a-4d04-a2b9-db2513e28cf5","resolution":{"observed_at":"2026-05-13T15:13:25.446063Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f158ea52-0ed9-4639-827e-a161fd46d56c","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:36431de92483853fe6e510f10bac499d3f505a3778b7ef115eec4b866ecde88e","observation_id":"96c83534-083a-46b0-a5f4-c2bbba8428bb","resolution":{"observed_at":"2026-05-13T15:13:25.450297Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"36a1f6da-def9-489f-92de-1af5021414ee","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:f4cacdcd4debdbed294b3b468a18dcee4c7d326fbf9ed21ba8596f8c9b0ff0a8","observation_id":"97c5cc0b-79c7-405c-9704-04de5d1b26d2","resolution":{"observed_at":"2026-05-13T15:13:25.454143Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"58e2f3d3-1e0b-4f57-a25c-af12561b2ffc","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:33dfc3fb3606a381a9bb2c047d02dfcf38ab974b97b3785374df7bc0d5408885","observation_id":"75062dbc-d0a3-4fc1-9764-fc9711c1425f","resolution":{"observed_at":"2026-05-13T15:13:25.458259Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ab53b5c2-10c0-4be9-a539-bef4500bf0bb","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:20b7dbd3b9c36891a00b8513bf34c75a1e4034cf089f8693d25c5a992a6628a0","observation_id":"0ee6e5ad-c4cc-4d44-a8c6-2b093683ecde","resolution":{"observed_at":"2026-05-13T15:13:25.462987Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"376dbf37-5265-4b43-bd3d-c87033c0edc1","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:ed52c62bde6664868e551a07718f20f98c07d8c50b44fdc3e9aa2aff3c8754a6","observation_id":"2b302ad6-b485-4af5-ac58-48afc5f9b1d1","resolution":{"observed_at":"2026-05-13T15:13:25.466618Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5dc7c59c-5cd5-4b5c-b99a-b6d708917e0d","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:bd4f01aa19597aecc6a2369e575e08b1da0046471d9b90283b72d7ec2db3a6c0","observation_id":"98978a21-05da-453c-bb0f-3ca2679f8132","resolution":{"observed_at":"2026-05-13T15:13:25.470473Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c11729d9-19a2-490c-8826-078e8e420d36","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:5588b6079cbb34f41a0d1b26368de196410cc436179d0f9960ec75c19786edb3","observation_id":"6f7cd57b-7194-449b-b645-da23a8cee567","resolution":{"observed_at":"2026-05-13T15:13:25.474517Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ae099845-a014-4ecd-bd61-bbe812b6f212","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:9bfc929c421131ad13489ff0a688ec01012f1ed425a0560d91382a4c1933b6f9","observation_id":"48c6805a-20b5-47f7-b74c-b5b73aac8339","resolution":{"observed_at":"2026-05-13T15:13:25.478159Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"These are just a few ideas to get you started","venue":null,"work_id":"aabc46be-d81e-455a-a7a8-8b3ed1766fda","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:709d745483a3101db324def56ac4a9eea6069fef7b6a6d0e2e82f88c7f80b02f","observation_id":"795c00ba-9007-40af-a9c1-191090fe20ad","resolution":{"observed_at":"2026-05-13T15:13:25.481889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"12b604ed-fff4-49d1-ad5d-1fdfc415151f","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:e4c35a75c3dcb8ae17f9b52bcd8f4c74a0587beaa304d61d46647fcc564e7665","observation_id":"068acaa6-f5a5-4991-b206-0b2471cd8873","resolution":{"observed_at":"2026-05-13T15:13:25.485104Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d5b1ba16-21c1-4684-a5ae-552d59ca6e38","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:294b3a71369c2a41f68b640a95b446aaea56d5faededacf0cb97c7710cf567ec","observation_id":"8a6f4ef4-59b7-46ed-9c85-7658b832cefe","resolution":{"observed_at":"2026-05-13T15:13:25.488773Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e2bc19eb-7754-4218-8614-ab7fa2de475f","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:93498bf205cd904690a5d6b57821998c59d37cc7b7834e5ec1a86b8c5008e2d2","observation_id":"f170670c-4834-46c4-a5b3-9e3fda7ec112","resolution":{"observed_at":"2026-05-13T15:13:25.492031Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cef202d2-afc0-4bff-b1fb-f91857958dd1","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:c14bc95951c4b3ddd032f2a06d4d65ca6583f898cd7e0a36346adde4da224437","observation_id":"1d8eb5cf-b8d5-4c00-a1cc-59afb239712c","resolution":{"observed_at":"2026-05-13T15:13:25.495280Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"My final verdict is tie: [[A=B]]","venue":null,"work_id":"6d27adff-092d-4c85-9c8d-6f171b71184e","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:8667e7782999509839ffd75283d1d92b60c697bea3757e729b475589f1f453e2","observation_id":"28a77312-fa14-4b52-bc9f-4afbf4aefaf6","resolution":{"observed_at":"2026-05-13T15:13:25.498802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"27 Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference - Continuously gather and incorporate customer feedback into the product development process","venue":null,"work_id":"a70c9e5b-d9cf-4d14-bd20-08a3cd54129a","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:5ff583524baa4a73dd139a61dd4002fe9a0624dc0c06dfa21fcc6cfe91f66641","observation_id":"3691fdaf-2448-44c4-b6e1-5f885dce0214","resolution":{"observed_at":"2026-05-13T15:13:25.503334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Align your product’s features and capabilities with its value proposition to ensure it meets the expectations of your target audience","venue":null,"work_id":"9d6645df-75b2-4336-9730-3203aac02818","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:a7f4de5cf5756301cf6fb0bee7729291a6ba3daeb17fc165d2b4c4553a28b04a","observation_id":"c7de7b00-0f13-463c-8e3d-9ef264470775","resolution":{"observed_at":"2026-05-13T15:13:25.507261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Validate assumptions and hypotheses through experimentation and user testing","venue":null,"work_id":"0f5fd2e2-7089-475e-bd85-30a5ee20ede9","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:5996cd43d56bef8dfe2b653cd0e00909ea454e3519c7c4e6955bdc4e53ae5377","observation_id":"fab17ebb-364f-4e64-bc43-81c393a82421","resolution":{"observed_at":"2026-05-13T15:13:25.511278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Be open to pivoting or making significant changes based on feedback and market response","venue":null,"work_id":"3428761d-5112-428b-96bf-8f7fbfaa91de","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:96d0f6e7697e93e6af97235b9d1ee18acd7695685d55df816fe02ff8b820dd43","observation_id":"691382cd-cec5-41b0-995d-813bb4a3d71c","resolution":{"observed_at":"2026-05-13T15:13:25.514912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Establish key performance indicators (KPIs) to measure the success of the product and track progress over time","venue":null,"work_id":"b01c9821-e979-4e8c-9394-163f2b114ec0","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:b9a51c47471d257b85a543fcf18255b8fd5cd4f40da176c0df9ab709263d0dd2","observation_id":"736c5688-4e45-44e9-b9f2-d1741812e124","resolution":{"observed_at":"2026-05-13T15:13:25.518950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Founders must be obsessed with their customers and be willing to put in the effort to understand their needs","venue":null,"work_id":"48c78154-c522-4496-ade2-a523ba4a6079","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:1527c3192025ee82ba5215ce143337bff3b3dfae96d961366ad65181eb710b45","observation_id":"a1938a07-32ef-46e6-8c6d-115900e32fd5","resolution":{"observed_at":"2026-05-13T15:13:25.522664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Founders must be willing to try new things, test hypotheses, and iterate on their product based on customer feedback","venue":null,"work_id":"c620ee1a-8154-40e5-ba31-d11985d0333f","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:34302c3c8c222de2519be1dcfc13d056d21b5731fba12adddd9479962bf45246","observation_id":"ff7b0c08-f83e-4cde-944c-007c14c6514d","resolution":{"observed_at":"2026-05-13T15:13:25.526414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Founders must be able to identify and prioritize the most important features and functionality that deliver the most value to their customers","venue":null,"work_id":"9994e1c9-aa4c-4803-b8c1-6b1b62323558","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:2e5972374aed59a8d4fb0c752ea01f2b4dcd1f5e01fc4fb3c05a61aca19d1733","observation_id":"fb6c5f2a-218f-4afc-b89c-3f40e719a856","resolution":{"observed_at":"2026-05-13T15:13:25.530386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Founders must be able to work effectively with these teams to develop a product that meets customer needs","venue":null,"work_id":"8d36434c-c228-4332-83b3-5e184d39fcdd","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:4a0619fc9f30b50cd840f0d8135de958e750c35ffd030b4f05ae37db6b6712c1","observation_id":"a053b572-8a53-4e6f-9303-f719c5f490bd","resolution":{"observed_at":"2026-05-13T15:13:25.536031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This includes analyzing customer feedback, usage data, and other metrics to inform product development","venue":null,"work_id":"57f33d1b-0234-48d6-a650-d5f6d3a22b4e","year":null},"citing_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T15:13:25.320641Z"},"links":{"citing_paper":"/paper/2403.04132"},"observation_digest":"sha256:122ae3f0c49e46a0ee675e990dbaef085e1ccf854c1f90b84bb000fbccf574e9","observation_id":"cf8ed48f-61ae-466a-acff-af8de76fc54e","resolution":{"observed_at":"2026-05-13T15:13:25.540637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":31,"verified_exact":1,"verified_fuzzy":15},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 93 inbound Pith citation observations for arXiv:2403.04132."}