{"as_of":"2026-08-06T05:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7db4f6fc6b1106a5db0194bf787c2413a53e1a0550996ec43960611784759635","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":31,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T14:21:48.754749Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T01:56:41.068880Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2408.15549","last_updated":"2026-04-17T16:47:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-28T05:53:46Z","title":"WildFeedback: Aligning LLMs With In-situ User Interactions And Feedback","version":4},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-23T22:37:43.230753Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2408.15549"},"observation_digest":"sha256:1e06e1928262e82e2b2a80546dc3700d9d963e28f7fad7406e135febcdd61f6d","observation_id":"333df622-23a6-4a25-b5b4-a4682dc55436","resolution":{"observed_at":"2026-05-23T22:38:32.538349Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2412.05579","last_updated":"2024-12-10T05:49:12Z","snapshot_observed_at":"2026-07-31T01:42:39.468673Z","submitted_at":"2024-12-07T08:07:24Z","title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","version":2},"reference_index":143,"source":"pdf_text","source_observed_at":"2026-05-11T23:08:34.312466Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2412.05579"},"observation_digest":"sha256:a7cf90bfa24c633fb9304fb056e5598c7ce3f9eafff4780c8b74ac6e264ca5ce","observation_id":"a03e328e-ab9a-4da2-823a-1dddb6be934a","resolution":{"observed_at":"2026-05-11T23:08:37.233353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2506.15787","last_updated":"2026-05-11T09:25:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-18T18:10:30Z","title":"SLR: Automated Synthesis for Scalable Logical Reasoning","version":6},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-19T08:47:17.008113Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2506.15787"},"observation_digest":"sha256:90bbe67e33d4d0d60d3d448cb00fd2a6a47d95518867ec3bdc8d323e1ccefa40","observation_id":"2c261279-5211-4122-8ebf-592572d9f32b","resolution":{"observed_at":"2026-05-19T08:52:13.798019Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-05T14:21:48.754749Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02594","last_updated":"2026-07-27T13:01:03Z","snapshot_observed_at":"2026-08-05T14:21:46.209025Z","submitted_at":"2025-08-29T09:51:41Z","title":"OpenAIs HealthBench in Action: Evaluating an LLM-Based Medical Assistant on Realistic Clinical Queries","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T14:21:48.754749Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2509.02594"},"observation_digest":"sha256:9fba3d95bd60d27a0db355e68c92b6275c980c603254681f3626762dfc132f00","observation_id":"cd8a3431-2352-4886-9242-4b74c792ef2d","resolution":{"observed_at":"2026-08-05T14:21:48.754749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2509.11206","last_updated":"2026-04-20T05:43:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-14T10:24:13Z","title":"Evalet: Evaluating Large Language Models through Functional Fragmentation","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-18T16:57:25.259866Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2509.11206"},"observation_digest":"sha256:de9cfb9eea790e7f8fd728973192279cdeee318a683d75f003082f820e5aca04","observation_id":"430098f2-be50-4b18-a573-9a7fa46550ed","resolution":{"observed_at":"2026-05-18T17:01:40.016367Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-04T11:34:16.004834Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.04033","last_updated":"2026-06-22T22:02:33Z","snapshot_observed_at":"2026-08-04T11:33:58.024115Z","submitted_at":"2025-10-05T04:52:26Z","title":"A global log for medical AI","version":2},"reference_index":127,"source":"pdf_text","source_observed_at":"2026-08-04T11:34:16.004834Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2510.04033"},"observation_digest":"sha256:26d77c38e7576162cb0d369c7b353ec2487d72135986801b3446d95089679cd9","observation_id":"9365063a-f02d-4276-8ad2-93eabb46581e","resolution":{"observed_at":"2026-08-04T11:34:16.004834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-04T09:22:43.139966Z","title":"Wild- bench: Benchmarking llms with challenging tasks from real users in the wild.arXiv preprint arXiv:2406.04770,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.15859","last_updated":"2026-05-29T10:51:31Z","snapshot_observed_at":"2026-08-05T08:56:19.289886Z","submitted_at":"2025-10-17T17:51:28Z","title":"InfiMed-ORBIT: Aligning LLMs on Open-Ended Complex Tasks via Rubric-Based Incremental Training","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T09:22:43.139966Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2510.15859"},"observation_digest":"sha256:59f672b80a12a0feca76e8b0ce5b95fdcdf8f800653a5c3d0dd55176c91307ec","observation_id":"fca7aa40-b80c-4670-830c-3d39af3f28e3","resolution":{"observed_at":"2026-08-04T09:22:43.139966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2601.08584","last_updated":"2026-01-13T14:06:03Z","snapshot_observed_at":"2026-08-04T17:10:53.354037Z","submitted_at":"2026-01-13T14:06:03Z","title":"Ministral 3","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-14T19:12:24.627033Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2601.08584"},"observation_digest":"sha256:bbccd599f152719ca6b73b60ea9a61b6d7d921aa2c4c68237d37eece17ee4019","observation_id":"966c419a-95d0-47d3-94f5-173d34f5809f","resolution":{"observed_at":"2026-05-14T19:12:24.743284Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2603.06003","last_updated":"2026-04-11T04:36:36Z","snapshot_observed_at":"2026-07-06T22:48:04.438669Z","submitted_at":"2026-03-06T08:02:58Z","title":"EvoESAP: Non-Uniform Expert Pruning for Sparse MoE","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T14:34:48.524592Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2603.06003"},"observation_digest":"sha256:97aa543846237a36de5fbecddefd1315680bcddf476691e87045de28f932daf8","observation_id":"2e57e716-744a-4c35-a42c-965a345511b7","resolution":{"observed_at":"2026-05-15T14:35:55.582310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-02T23:09:26.247244Z","title":"URL https: //arxiv.org/abs/2406.04770","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.10002","last_updated":"2026-07-02T20:32:23Z","snapshot_observed_at":"2026-08-02T23:09:24.567871Z","submitted_at":"2026-02-16T14:24:36Z","title":"SpreadsheetArena: Decomposing Preference in LLM Generation of Spreadsheet Workbooks","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-02T23:09:26.247244Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2603.10002"},"observation_digest":"sha256:ddc722de56324d93c71e1f493e6d29b22f19bec4d4bfc590fd9b1e60b803a683","observation_id":"486f3654-1f72-4add-af90-0b07c8808378","resolution":{"observed_at":"2026-08-02T23:09:26.247244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-02T23:09:26.124170Z","title":"Lambert, N., Morrison, J., Pyatkin, V ., Huang, S., Ivison, H., Brahman, F., Miranda, L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.10002","last_updated":"2026-07-02T20:32:23Z","snapshot_observed_at":"2026-08-02T23:09:24.567871Z","submitted_at":"2026-02-16T14:24:36Z","title":"SpreadsheetArena: Decomposing Preference in LLM Generation of Spreadsheet Workbooks","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-02T23:09:26.124170Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2603.10002"},"observation_digest":"sha256:29eebff76d576bf6be124a3bb5d37faccc38cfad221f8e373f6afa1b7654c882","observation_id":"5d1de721-441c-48f9-97e7-bb3e31d73ef4","resolution":{"observed_at":"2026-08-02T23:09:26.124170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2603.27977","last_updated":"2026-05-10T14:55:46Z","snapshot_observed_at":"2026-07-31T06:59:54.210184Z","submitted_at":"2026-03-30T02:54:48Z","title":"SARL: Label-Free Reinforcement Learning by Rewarding Reasoning Topology","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-14T22:20:42.183878Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2603.27977"},"observation_digest":"sha256:54c22d0119d24c11aae8ca652b583c48012514b28703962c1fd1c201394bc01f","observation_id":"a3f08183-652c-43b0-ab00-651e72041cec","resolution":{"observed_at":"2026-05-14T22:23:03.627283Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2604.06996","last_updated":"2026-08-03T11:57:08Z","snapshot_observed_at":"2026-08-06T04:41:25.258983Z","submitted_at":"2026-04-08T12:13:53Z","title":"Self-Preference Bias in Rubric-Based Evaluation of Large Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:18:19.955943Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2604.06996"},"observation_digest":"sha256:55ecbc5d16f3ab5627ae4854cad3410bea8b82a9e5c44ab1dfaa271053f8d590","observation_id":"9d4c6217-117d-4b38-a557-c4b0b1dbf71d","resolution":{"observed_at":"2026-05-11T00:50:50.492065Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-02T16:40:55.432902Z","title":"Wildbench: Benchmark- ing llms with challenging tasks from real users in the wild.arXiv preprint arXiv:2406.04770,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.06996","last_updated":"2026-08-03T11:57:08Z","snapshot_observed_at":"2026-08-06T04:41:25.258983Z","submitted_at":"2026-04-08T12:13:53Z","title":"Self-Preference Bias in Rubric-Based Evaluation of Large Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T16:40:55.432902Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2604.06996"},"observation_digest":"sha256:6279eff492273347e8462cab6a745c3494d1f0e8240ff9326015d8ad72e7eace","observation_id":"5c809790-9671-44a8-8359-bf613eaf5d8f","resolution":{"observed_at":"2026-08-02T16:40:55.432902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-04T05:36:43.341882Z","title":"Wildbench: Benchmark- ing llms with challenging tasks from real users in the wild.arXiv preprint arXiv:2406.04770,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.06996","last_updated":"2026-08-03T11:57:08Z","snapshot_observed_at":"2026-08-06T04:41:25.258983Z","submitted_at":"2026-04-08T12:13:53Z","title":"Self-Preference Bias in Rubric-Based Evaluation of Large Language Models","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T05:36:43.341882Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2604.06996"},"observation_digest":"sha256:f23f909d6ff0d8a71bc182bc80c6b53f401527f7c9143f8e282a89b9d4938eea","observation_id":"4ddca33e-c1f3-4791-a5b2-43df81a94cae","resolution":{"observed_at":"2026-08-04T05:36:43.341882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2604.07837","last_updated":"2026-04-09T05:37:22Z","snapshot_observed_at":"2026-07-06T22:57:05.146373Z","submitted_at":"2026-04-09T05:37:22Z","title":"SPARD: Self-Paced Curriculum for RL Alignment via Integrating Reward Dynamics and Data Utility","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T17:42:57.745612Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2604.07837"},"observation_digest":"sha256:6383ec71bda44bea6985f0a84dfba6aea8ea33aaf01e668754008f371a0c89e8","observation_id":"e6301820-d308-4e8e-a821-8f120c81072c","resolution":{"observed_at":"2026-05-11T06:15:58.269538Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2604.14585","last_updated":"2026-05-27T06:20:29Z","snapshot_observed_at":"2026-07-12T20:11:17.665257Z","submitted_at":"2026-04-16T03:23:46Z","title":"Prompt Optimization Is a Coin Flip: Diagnosing When It Helps in Compound AI Systems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T11:18:09.127345Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2604.14585"},"observation_digest":"sha256:8bdfa7855ea95e1025d2c386673d50e27337920371e4a40bbab83253c084c889","observation_id":"2d604029-404f-459e-b770-e98a57a70e77","resolution":{"observed_at":"2026-05-10T11:20:10.373403Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.02209","last_updated":"2026-05-04T04:17:02Z","snapshot_observed_at":"2026-07-06T23:15:18.048504Z","submitted_at":"2026-05-04T04:17:02Z","title":"Submodular Benchmark Selection","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-08T19:23:51.649257Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.02209"},"observation_digest":"sha256:dd5003b7cfb06e5f17c9fce9accac81e02eea8eff853b503281bc96b96765f0c","observation_id":"a7e803ed-fad8-4877-b38d-803aab63ef84","resolution":{"observed_at":"2026-05-09T05:55:30.309294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.04454","last_updated":"2026-05-06T03:28:30Z","snapshot_observed_at":"2026-08-04T06:38:56.344758Z","submitted_at":"2026-05-06T03:28:30Z","title":"Deployment-Relevant Alignment Cannot Be Inferred from Model-Level Evaluation Alone","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-08T18:11:42.971428Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.04454"},"observation_digest":"sha256:41c16abb337d69e75073c7a9bc82dcfdff628b3694486f840323621759cbbfbb","observation_id":"d3401c76-3a18-4044-9471-62c4fc37d666","resolution":{"observed_at":"2026-05-09T06:40:43.319500Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.09228","last_updated":"2026-05-09T23:56:04Z","snapshot_observed_at":"2026-07-06T23:21:21.560069Z","submitted_at":"2026-05-09T23:56:04Z","title":"ProactBench: Beyond What The User Asked For","version":1},"reference_index":123,"source":"arxiv_source","source_observed_at":"2026-05-12T02:14:01.145443Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.09228"},"observation_digest":"sha256:431a6670397ec5d5689e2f2e3789e7621f7de38f2d20fc5119402453f73c75e8","observation_id":"6f66d758-382c-45a7-9032-dbd91a7ef947","resolution":{"observed_at":"2026-05-12T02:16:16.058678Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.09808","last_updated":"2026-05-10T23:06:24Z","snapshot_observed_at":"2026-08-02T12:27:41.948284Z","submitted_at":"2026-05-10T23:06:24Z","title":"Quantifying the Utility of User Simulators for Building Collaborative LLM Assistants","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-12T02:28:13.317630Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.09808"},"observation_digest":"sha256:cd131651d63799a803ab4c17d430cc010aa943a5882c905dc58a5c71e11bf0d2","observation_id":"d068d95e-3fba-4153-979b-5947cc41ca9e","resolution":{"observed_at":"2026-05-12T07:36:48.587483Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.18721","last_updated":"2026-05-21T04:23:01Z","snapshot_observed_at":"2026-07-06T23:29:33.702647Z","submitted_at":"2026-05-18T17:50:27Z","title":"General Preference Reinforcement Learning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-20T12:43:56.522345Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.18721"},"observation_digest":"sha256:a4f88445734860a6ff858d87c8f10542f1d1e811d409914f699930f78e5c4cbc","observation_id":"0d5e2a77-4f95-4d5d-bfb0-53968831b645","resolution":{"observed_at":"2026-05-20T12:48:17.744570Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.18721","last_updated":"2026-05-21T04:23:01Z","snapshot_observed_at":"2026-07-06T23:29:33.702647Z","submitted_at":"2026-05-18T17:50:27Z","title":"General Preference Reinforcement Learning","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-21T07:50:00.963837Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.18721"},"observation_digest":"sha256:859a29a0ed2e21580d362b6a15e0f02bad9076d793244a2da687443afec8252d","observation_id":"c7de08d9-9862-4ecd-9fca-5c26f3d31a63","resolution":{"observed_at":"2026-05-21T07:54:02.863252Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.18721","last_updated":"2026-05-21T04:23:01Z","snapshot_observed_at":"2026-07-06T23:29:33.702647Z","submitted_at":"2026-05-18T17:50:27Z","title":"General Preference Reinforcement Learning","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T09:24:39.228616Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.18721"},"observation_digest":"sha256:361e6f2a22028e07923ff5626f7cffd9e89330c002f54a3e771cf77631fcec3e","observation_id":"22e02d68-cf31-4d4a-870d-0d1869249cd8","resolution":{"observed_at":"2026-05-22T09:24:45.688249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.20520","last_updated":"2026-05-19T21:42:32Z","snapshot_observed_at":"2026-08-03T02:30:24.579386Z","submitted_at":"2026-05-19T21:42:32Z","title":"Open-World Evaluations for Measuring Frontier AI Capabilities","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-21T06:38:51.427985Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.20520"},"observation_digest":"sha256:d545a03f6f5a313742746193c64670d0afdde7111a459ca36b7fc9c38db395be","observation_id":"f3e2b446-5cc0-4b78-ba83-21ff1527939a","resolution":{"observed_at":"2026-05-21T06:39:43.678968Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2605.21086","last_updated":"2026-05-20T12:21:15Z","snapshot_observed_at":"2026-07-06T23:31:37.061253Z","submitted_at":"2026-05-20T12:21:15Z","title":"LoCar: Localization-Aware Evaluation of In-Vehicle Assistants through Fine-Grained Sociolinguistic Control","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-21T05:02:31.689194Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2605.21086"},"observation_digest":"sha256:4c48bf89b88dd046db2f58a78d391faadd8f684556341faa846cd62cb7c4825d","observation_id":"383d3279-6317-48ae-88c4-f430615758c6","resolution":{"observed_at":"2026-05-21T05:03:57.894432Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":"2406.04770","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-07-10T01:56:41.068880Z","title":"Wildbench: Benchmarking llms with challenging tasks from real users in the wild, 2024 a","venue":"cs.CL","work_id":"7018fa84-745f-452a-aaa9-075fdc9a8718","year":2024},"citing_paper":{"arxiv_id":"2607.08758","last_updated":"2026-07-09T17:55:53Z","snapshot_observed_at":"2026-07-12T23:18:59.558861Z","submitted_at":"2026-07-09T17:55:53Z","title":"Ideas Have Genomes: Benchmarking Scientific Lineage Reasoning and Lineage-Grounded Idea Generation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-10T01:51:14.922841Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2607.08758"},"observation_digest":"sha256:29d27c62e6bbe5092339cc64a0130da2adab570d868991e8cdab91188d5a4965","observation_id":"d634685b-3815-4c48-8c0d-7db91b7bd372","resolution":{"observed_at":"2026-07-10T01:56:41.070852Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-01T21:31:54.729032Z","title":"doi:10.48550/arXiv.2406.04770 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16057","last_updated":"2026-08-03T18:41:38Z","snapshot_observed_at":"2026-08-06T04:50:40.300177Z","submitted_at":"2026-07-17T15:34:51Z","title":"Frontier AI performance across the business disciplines: a case-grounded benchmark of knowledge work and analytical reasoning","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-01T21:31:54.729032Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2607.16057"},"observation_digest":"sha256:77ad10a9e7861abaa05e7f5dcee46a0f3788531f2201d7f46176e03f7ac5627e","observation_id":"d1bae9ca-a40d-4ca4-8d85-52e438fb6a62","resolution":{"observed_at":"2026-08-01T21:31:54.729032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-03T00:49:50.931706Z","title":"doi:10.48550/arXiv.2406.04770 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16057","last_updated":"2026-08-03T18:41:38Z","snapshot_observed_at":"2026-08-06T04:50:40.300177Z","submitted_at":"2026-07-17T15:34:51Z","title":"Frontier AI performance across the business disciplines: a case-grounded benchmark of knowledge work and analytical reasoning","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-03T00:49:50.931706Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2607.16057"},"observation_digest":"sha256:5d924231c97ff0487f7dbc41b98d6f2bed5453fb41e59f2be91649e2338ae5c3","observation_id":"45081487-b398-4748-b9b5-2dd1067f557e","resolution":{"observed_at":"2026-08-03T00:49:50.931706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-01T15:29:24.341223Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18438","last_updated":"2026-07-20T18:46:17Z","snapshot_observed_at":"2026-08-04T02:52:47.040983Z","submitted_at":"2026-07-20T18:46:17Z","title":"Relay-Bench: Evaluating LLMs on Multi-Domain Reasoning Chains","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T15:29:24.341223Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2607.18438"},"observation_digest":"sha256:af3e423ae1d82b9969070f34ce38e4487cc14bc859e0fd4f3acdb2e142b3dd60","observation_id":"79d822aa-4ddb-4df3-bd1e-ec4c29589558","resolution":{"observed_at":"2026-08-01T15:29:24.341223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04770","snapshot_observed_at":"2026-08-02T13:58:46.097954Z","title":"Y.et al.Wildbench: Benchmarking llms with challenging tasks from real users in the wild (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.20454","last_updated":"2026-05-14T21:06:34Z","snapshot_observed_at":"2026-08-02T13:58:43.973692Z","submitted_at":"2026-05-14T21:06:34Z","title":"Response drift across frontier large language models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T13:58:46.097954Z"},"links":{"cited_paper":"/paper/2406.04770","citing_paper":"/paper/2607.20454"},"observation_digest":"sha256:7426863f75940fb88bff41c72cb2e098d89ef757c7364f5210dc1af23e20e976","observation_id":"b4c8ce04-d87e-442f-8b76-86ac69f15ed4","resolution":{"observed_at":"2026-08-02T13:58:46.097954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.04770/citation-record","integrity":"/paper/2406.04770/integrity","json":"/paper/2406.04770/citation-record.json","paper":"/paper/2406.04770"},"outbound":[],"paper":{"arxiv_id":"2406.04770","last_updated":"2024-10-05T22:39:51Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T09:15:44Z","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 31 inbound Pith citation observations for arXiv:2406.04770."}