{"as_of":"2026-08-06T18:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c8021c2f23a3d49d35405723dec0423fe298c6978cedd306ea30526d480d7622","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-11T15:51:04.674346Z","state":"measured"},{"denominator":154,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":154,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":163,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:54:17.819861Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":13,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-15T04:48:26.303240Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2406.19314"},"observation_digest":"sha256:0a0d72b631be74f23e07048923c316fcc93b2e3e06158216a83e269db61942d2","observation_id":"5589a9fd-58b4-44af-ab06-1c4b4e693fbf","resolution":{"observed_at":"2026-05-15T04:48:26.363473Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2408.15339","last_updated":"2026-05-07T20:32:48Z","snapshot_observed_at":"2026-08-06T15:22:00.046544Z","submitted_at":"2024-08-27T18:04:07Z","title":"UNA: A Unified Supervised Framework for Efficient LLM Alignment Across Feedback Types","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-23T21:22:36.970101Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2408.15339"},"observation_digest":"sha256:2129ca4e8e5418d45d67eec74ed5ea1ca8e37a8ec032fb76a0545ef31d198c19","observation_id":"cbd8afec-9ea4-41ba-b087-9325b32ea148","resolution":{"observed_at":"2026-05-23T21:23:27.390797Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2409.02231","last_updated":"2026-04-10T22:29:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-03T18:59:20Z","title":"SmileyLlama: Modifying Large Language Models for Directed Chemical Space Exploration","version":5},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-23T21:16:39.979945Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2409.02231"},"observation_digest":"sha256:3360f971ab9bedf287725002893088c06205a9eaf2f7b67e48f1e067394b23d2","observation_id":"d5817a28-27f2-4682-ba99-49f6ee4d2de2","resolution":{"observed_at":"2026-05-23T21:18:26.949262Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2409.02813","last_updated":"2025-05-22T08:22:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-04T15:31:26Z","title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","version":3},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-14T00:51:48.163349Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2409.02813"},"observation_digest":"sha256:65653dcd5ab6ac7dd10c2efbb7e2bb7211fd4f673af219ef9676ec6893aa2f88","observation_id":"fb473fcc-da64-407d-becc-3038ddc8d4ef","resolution":{"observed_at":"2026-05-14T00:51:48.416467Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2410.17196","last_updated":"2024-12-11T15:45:21Z","snapshot_observed_at":"2026-07-06T19:37:56.214143Z","submitted_at":"2024-10-22T17:15:20Z","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","version":3},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-17T00:50:13.841689Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2410.17196"},"observation_digest":"sha256:de27973e077ae9a2b599c95539477b669d8e292e57706da7c226bff3279dfc76","observation_id":"31450546-1025-41f3-a1d7-210457e2cc3b","resolution":{"observed_at":"2026-05-17T00:50:14.022353Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2412.14590","last_updated":"2026-04-22T15:43:48Z","snapshot_observed_at":"2026-08-03T01:50:37.847212Z","submitted_at":"2024-12-19T07:15:15Z","title":"MixLLM: LLM Quantization with Global Mixed-precision between Output-features and Highly-efficient System Design","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-23T06:56:51.829741Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2412.14590"},"observation_digest":"sha256:1e67f345dcc820545686cd96836d26c57a21873b14862caa518f6d8c4011c4fc","observation_id":"14bf5fc2-f624-4d6f-8bad-89df8cbd4141","resolution":{"observed_at":"2026-05-23T06:57:40.280362Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-23T06:25:00.376073Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2412.15115"},"observation_digest":"sha256:62b11c1c909a5979b9079f2f33f0259412ccac83df890e8fe442919c33cdf87c","observation_id":"8718085d-9a5e-4b4f-a63a-211e2601203f","resolution":{"observed_at":"2026-05-23T06:25:27.932902Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2412.18925","last_updated":"2024-12-25T15:12:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-25T15:12:34Z","title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-15T12:36:50.060335Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2412.18925"},"observation_digest":"sha256:f234106bf683cf3fb15795c0ed183da1fe96da6152e915b618b9ac900ec7cd6f","observation_id":"b9ffacf9-c1fd-4cf2-962c-989fdc273cb8","resolution":{"observed_at":"2026-05-15T12:36:50.187431Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2501.09686","last_updated":"2025-01-23T08:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T17:37:58Z","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","version":3},"reference_index":156,"source":"pdf_text","source_observed_at":"2026-05-15T21:20:59.128986Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2501.09686"},"observation_digest":"sha256:162c7c0451ca17406d96fd791a3183bf772d51ab808257346aa0f2a2566c88a2","observation_id":"f9a943b6-6e23-42b6-9f29-1c77c448da02","resolution":{"observed_at":"2026-05-15T21:20:59.341382Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2501.13826","last_updated":"2025-01-23T16:51:47Z","snapshot_observed_at":"2026-07-06T20:25:03.950783Z","submitted_at":"2025-01-23T16:51:47Z","title":"Video-MMMU: Evaluating Knowledge Acquisition from Multi-Discipline Professional Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-14T00:32:41.059558Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2501.13826"},"observation_digest":"sha256:8181c5a5ca2c97453d6905449efba783459b1dce0d11247186fd65a3c1713ea7","observation_id":"95cf7e11-30a1-4de7-bf0b-3c86ec5b9bdc","resolution":{"observed_at":"2026-05-14T00:32:41.308387Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T18:40:50.139345Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2501.14249"},"observation_digest":"sha256:afbda548b9021510206a7c1efd17f5a849af4a4b86ee502b05483360d5f5ac50","observation_id":"06000166-b00c-4927-90f0-36eb17c4098f","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2502.02737","last_updated":"2025-02-04T21:43:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-04T21:43:16Z","title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","version":1},"reference_index":237,"source":"arxiv_source","source_observed_at":"2026-05-13T17:30:02.803757Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2502.02737"},"observation_digest":"sha256:753d41bb3bfff9be377886ed4b8b315622c022b84e851b3a8e465c84cf61934b","observation_id":"cb0a87d3-9c22-46de-8c26-ea71076eb1c0","resolution":{"observed_at":"2026-05-13T17:30:03.042368Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2502.11089","last_updated":"2025-02-27T09:01:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-16T11:53:44Z","title":"Native Sparse Attention: Hardware-Aligned and Natively Trainable Sparse Attention","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-16T23:46:29.975858Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2502.11089"},"observation_digest":"sha256:6a2dbd529e93d84ce0a3e5fd5e147fcdde60e80e63599d4ea533d224094735b0","observation_id":"5a9aba33-b748-488a-9882-52af28f6a352","resolution":{"observed_at":"2026-05-16T23:46:30.121500Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2504.14945","last_updated":"2025-06-22T00:18:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-21T08:09:13Z","title":"Learning to Reason under Off-Policy Guidance","version":5},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T23:17:02.701393Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2504.14945"},"observation_digest":"sha256:e9a09ad002eb71956380417f3f3cb6d2727c7bc2de055bedf573b22e56e800ce","observation_id":"945cac13-4dbf-4c8b-bafd-ca751ce6d86b","resolution":{"observed_at":"2026-05-15T23:17:02.817507Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2504.16155","last_updated":"2026-05-07T09:59:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-22T17:52:04Z","title":"PRIMETIME : Limits of LLMs in Temporal Primitives","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-22T18:36:48.376877Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2504.16155"},"observation_digest":"sha256:0800d263d072ce7565f6f6dba36860c10eb9588710d53a2d0d7c0013f422d380","observation_id":"40de151a-8bb8-4919-b6aa-c15a3dee4675","resolution":{"observed_at":"2026-05-22T18:36:58.819218Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2504.21318","last_updated":"2025-04-30T05:05:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-30T05:05:09Z","title":"Phi-4-reasoning Technical Report","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-17T03:40:25.706499Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2504.21318"},"observation_digest":"sha256:c54e48c10d129042374da94b0956d9d81cac1a814026b26987d2533db72b6231","observation_id":"21a94ca9-336c-4565-b90e-f089ea37fe87","resolution":{"observed_at":"2026-05-17T03:40:25.847444Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-09T06:35:27.813995Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2505.09388"},"observation_digest":"sha256:8db18bafbe58d6a52fe47fff7d00d43df5b8795f1c30b5ca0a2e0b8773d326a2","observation_id":"add70447-8e26-4327-81e8-9795bd1463e3","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T17:54:17.819861Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09662","last_updated":"2025-07-13T14:51:59Z","snapshot_observed_at":"2026-08-06T17:48:05.466436Z","submitted_at":"2025-07-13T14:51:59Z","title":"Towards Concise and Adaptive Thinking in Large Reasoning Models: A Survey","version":1},"reference_index":195,"source":"arxiv_source","source_observed_at":"2026-08-06T17:54:17.819861Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.09662"},"observation_digest":"sha256:5ef9e61a948ff0c52e70ae47b3d1612a3c8f32310cdc24aad0feeea1fd27bf82","observation_id":"68aa4485-0168-4fc3-aac7-990f6a78d6e4","resolution":{"observed_at":"2026-08-06T17:54:17.819861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T17:52:01.680504Z","title":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E., Le, Q., and Zhou, D","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.10616","last_updated":"2025-07-25T11:09:53Z","snapshot_observed_at":"2026-08-06T17:46:01.198994Z","submitted_at":"2025-07-13T19:04:17Z","title":"Scalpel vs. Hammer: GRPO Amplifies Existing Capabilities, SFT Replaces Them","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:52:01.680504Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.10616"},"observation_digest":"sha256:c8f910c4a6ce65acd9e3979b9d74738cdd8bf4844fcd7dc7c737875861dc3a13","observation_id":"600ea502-3724-4313-bc88-ff89b642cfd6","resolution":{"observed_at":"2026-08-06T17:52:01.680504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T17:16:54.629011Z","title":"arXiv:2406.01574 [cs.CL] https://arxiv.org/abs/ 2406.01574 Zhilin Yang, Peng Qi, Saizheng Zhang, Yoshua Bengio, William Cohen, Ruslan Salakhutdinov, and Christopher D","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11371","last_updated":"2025-07-15T14:44:29Z","snapshot_observed_at":"2026-08-06T17:07:04.231936Z","submitted_at":"2025-07-15T14:44:29Z","title":"Step-wise Policy for Rare-tool Knowledge (SPaRK): Offline RL that Drives Diverse Tool Use in LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:16:54.629011Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.11371"},"observation_digest":"sha256:bdbaf88f060b6fd56d6d569a4d63153eee13d833edb17b609b10ef7384550742","observation_id":"d6989800-c228-4a3f-b99c-6821732c4721","resolution":{"observed_at":"2026-08-06T17:16:54.629011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T16:46:02.127936Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12806","last_updated":"2025-08-01T22:37:16Z","snapshot_observed_at":"2026-08-06T16:35:50.892428Z","submitted_at":"2025-07-17T05:46:27Z","title":"MCPEval: Automatic MCP-based Deep Evaluation for AI Agent Models","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T16:46:02.127936Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.12806"},"observation_digest":"sha256:703bda94623a0aadfe85736e32a849c54d20009a1ba00b4d6ebed2ab3550a075","observation_id":"a242dc43-64ca-404a-a3e3-fadb42c1a74e","resolution":{"observed_at":"2026-08-06T16:46:02.127936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T16:14:06.701604Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14295","last_updated":"2025-08-22T16:49:10Z","snapshot_observed_at":"2026-08-06T15:57:18.677408Z","submitted_at":"2025-07-18T18:07:38Z","title":"A Simple \"Try Again\" Can Elicit Multi-Turn LLM Reasoning","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T16:14:06.701604Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.14295"},"observation_digest":"sha256:66c70900ebf08dafcf02ced8e0ff0056f78afad26b533562bb1bcc171214e14a","observation_id":"afae9fdf-48b2-4f3c-856e-656045a2e324","resolution":{"observed_at":"2026-08-06T16:14:06.701604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T15:53:17.628302Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14805","last_updated":"2025-07-20T03:51:13Z","snapshot_observed_at":"2026-08-06T15:44:53.955040Z","submitted_at":"2025-07-20T03:51:13Z","title":"Subliminal Learning: Language models transmit behavioral traits via hidden signals in data","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T15:53:17.628302Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.14805"},"observation_digest":"sha256:5313793902ec32ec9ec2acc771d4b1d55a756d3351eb50abfdff39f80dbc81a3","observation_id":"30a969c8-e2ab-4366-beb1-6645b26abb7b","resolution":{"observed_at":"2026-08-06T15:53:17.628302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T15:50:04.421502Z","title":"Mmlu-pro: A more robust and chal- lenging multi-task language understanding benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14928","last_updated":"2025-07-20T11:55:26Z","snapshot_observed_at":"2026-08-06T15:42:09.688711Z","submitted_at":"2025-07-20T11:55:26Z","title":"Byzantine-Robust Decentralized Coordination of LLM Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T15:50:04.421502Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.14928"},"observation_digest":"sha256:0da6495a0f16e4f59d64e5693c58003606e9efe24bc5528c70c72628968bad41","observation_id":"92f2e232-1f35-450f-b829-fa4cc44dcb17","resolution":{"observed_at":"2026-08-06T15:50:04.421502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T15:14:22.986497Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16534","last_updated":"2025-07-26T12:33:42Z","snapshot_observed_at":"2026-08-06T15:05:13.017243Z","submitted_at":"2025-07-22T12:44:38Z","title":"Frontier AI Risk Management Framework in Practice: A Risk Analysis Technical Report","version":2},"reference_index":2006,"source":"pdf_text","source_observed_at":"2026-08-06T15:14:22.986497Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.16534"},"observation_digest":"sha256:0f583dcf90902f45e96f250e653e499aff9dc7cb67c2a6792aebf4ec19399c20","observation_id":"118bef26-7bfa-44a8-bc95-f54a0330bc81","resolution":{"observed_at":"2026-08-06T15:14:22.986497Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T14:43:25.916904Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18013","last_updated":"2025-07-29T10:30:18Z","snapshot_observed_at":"2026-08-06T14:36:15.969801Z","submitted_at":"2025-07-24T01:00:48Z","title":"Technical Report of TeleChat2, TeleChat2.5 and T1","version":3},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T14:43:25.916904Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.18013"},"observation_digest":"sha256:907c74784bfb2353fed7f17eb44a4d74bdc5b8baa972987f6e4b2f72b0ba8446","observation_id":"97b4180d-1638-48b4-926a-3bd571758a57","resolution":{"observed_at":"2026-08-06T14:43:25.916904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2507.20534","last_updated":"2026-02-03T04:57:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-28T05:35:43Z","title":"Kimi K2: Open Agentic Intelligence","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-10T17:49:27.926646Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.20534"},"observation_digest":"sha256:01607fa35f514935a6f664149efbfa15ee7edbb08aaa1afcd0e1c07b284feeaa","observation_id":"5874fde9-dc86-446d-8155-58c5b371f3d9","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T11:44:05.184665Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.22448","last_updated":"2025-07-30T07:55:33Z","snapshot_observed_at":"2026-08-06T11:43:58.532083Z","submitted_at":"2025-07-30T07:55:33Z","title":"Falcon-H1: A Family of Hybrid-Head Language Models Redefining Efficiency and Performance","version":1},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-08-06T11:44:05.184665Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.22448"},"observation_digest":"sha256:62f94f7f74f2e1b08c858fa171a83deef112189250698854c109234005c86422","observation_id":"42eedb8e-00ca-43e7-96aa-caf9f14b1d17","resolution":{"observed_at":"2026-08-06T11:44:05.184665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T10:07:27.652516Z","title":"The correct answer is (insert answer here)","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.00614","last_updated":"2025-08-01T13:23:21Z","snapshot_observed_at":"2026-08-06T10:07:24.175444Z","submitted_at":"2025-08-01T13:23:21Z","title":"Prompting Science Report 3: I'll pay you or I'll kill you -- but will you care?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T10:07:27.652516Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2508.00614"},"observation_digest":"sha256:ab75998dc04922d560ec02e84315a0800d188d93a241039af758114cab6bd920","observation_id":"f71066b1-4e17-4f29-bf81-e29be213e7c3","resolution":{"observed_at":"2026-08-06T10:07:27.652516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-06T05:11:56.015039Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.02208","last_updated":"2025-08-05T14:01:00Z","snapshot_observed_at":"2026-08-06T05:11:49.194221Z","submitted_at":"2025-08-04T08:59:36Z","title":"Proof2Hybrid: Automatic Mathematical Benchmark Synthesis for Proof-Centric Problems","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T05:11:56.015039Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2508.02208"},"observation_digest":"sha256:444ccae9accfbcb05f8da63e111091ef0c91b9929aa55c3d51a84be54c71c79e","observation_id":"617e65e0-d88e-43a3-b3e7-c7ef7f86d107","resolution":{"observed_at":"2026-08-06T05:11:56.015039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T20:33:21.816380Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10349","last_updated":"2025-08-14T05:14:00Z","snapshot_observed_at":"2026-08-05T20:33:18.047532Z","submitted_at":"2025-08-14T05:14:00Z","title":"Flexible Personalized Split Federated Learning for On-Device Fine-Tuning of Foundation Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T20:33:21.816380Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2508.10349"},"observation_digest":"sha256:1fd82a5a2b3aaf131c08436d663825395cc61617f18520acb35a0d5734073c57","observation_id":"8d4f49e4-83bb-4472-a2bd-1350ffca6a77","resolution":{"observed_at":"2026-08-05T20:33:21.816380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2508.12851","last_updated":"2026-04-15T14:22:25Z","snapshot_observed_at":"2026-07-06T22:14:25.511807Z","submitted_at":"2025-08-18T11:41:17Z","title":"Accelerating Edge Inference for Distributed MoE Models with Latency-Optimized Expert Placement","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T22:52:39.416575Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2508.12851"},"observation_digest":"sha256:ee66e61f23250a42b9d5f41deb81cb46a461f592c476fbd470ea2914b13ac87d","observation_id":"33b53e28-96ef-4d18-8419-7549eb4b4098","resolution":{"observed_at":"2026-05-18T22:52:51.881624Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T17:26:28.929994Z","title":"MMLU-Pro : A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.16279","last_updated":"2025-08-22T10:35:56Z","snapshot_observed_at":"2026-08-06T11:01:05.912134Z","submitted_at":"2025-08-22T10:35:56Z","title":"AgentScope 1.0: A Developer-Centric Framework for Building Agentic Applications","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-05T17:26:28.929994Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2508.16279"},"observation_digest":"sha256:4cbb600712aa2ee46a49b12d587baffe62c3175b348ae007d0cc435fc4607f19","observation_id":"d27c7d98-936a-4bea-85de-777b107ba884","resolution":{"observed_at":"2026-08-05T17:26:28.929994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T16:32:54.640365Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.18255","last_updated":"2025-09-02T17:12:12Z","snapshot_observed_at":"2026-08-05T16:32:51.675767Z","submitted_at":"2025-08-25T17:45:06Z","title":"Hermes 4 Technical Report","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-05T16:32:54.640365Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2508.18255"},"observation_digest":"sha256:455c3d058a350c5c86f295aa9e1951401647ca94ec13c5ed56be1fbedfce2982","observation_id":"b19ccd90-2ba7-494b-959a-c7240342e329","resolution":{"observed_at":"2026-08-05T16:32:54.640365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":147,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:1a5191b1e269924b113f1b242b782b190b7b9bcb557f40ef2b6952f7eb1349e5","observation_id":"ec688596-32e5-417f-8af7-837f217f0b75","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T13:55:01.947238Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.05316","last_updated":"2026-06-05T13:49:14Z","snapshot_observed_at":"2026-08-05T13:54:56.598714Z","submitted_at":"2025-08-29T19:25:52Z","title":"Standard vs. Modular Sampling: Best Practices for Reliable LLM Unlearning","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:01.947238Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2509.05316"},"observation_digest":"sha256:73bd79daa1e3ece738771f86dbe67b97eec158aa375e185c500060782347e00e","observation_id":"3c90f828-096b-480c-bf77-958bc06d5451","resolution":{"observed_at":"2026-08-05T13:55:01.947238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-04T19:48:48.841252Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09097","last_updated":"2025-09-11T02:16:34Z","snapshot_observed_at":"2026-08-06T08:01:16.387173Z","submitted_at":"2025-09-11T02:16:34Z","title":"DP-FedLoRA: Privacy-Enhanced Federated Fine-Tuning for On-Device Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T19:48:48.841252Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2509.09097"},"observation_digest":"sha256:c119603bd80a819fcce6055d31ea8df8b0a56904943f9e967e1b7e32de93da34","observation_id":"c0a51aa7-b234-4598-a0dd-fdd48ef86b7f","resolution":{"observed_at":"2026-08-04T19:48:48.841252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-04T17:05:55.034000Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.11155","last_updated":"2025-09-14T08:20:48Z","snapshot_observed_at":"2026-08-04T17:05:50.271239Z","submitted_at":"2025-09-14T08:20:48Z","title":"AQUA: Attention via QUery mAgnitudes for Memory and Compute Efficient Inference in LLMs","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-04T17:05:55.034000Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2509.11155"},"observation_digest":"sha256:d94b4d98441995f095625851504e7469a1cf6a8c3429b2182e906e197a5eb3c6","observation_id":"41752cb9-bf29-4457-9506-43f60f78cadd","resolution":{"observed_at":"2026-08-04T17:05:55.034000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-04T07:57:25.863201Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.23497","last_updated":"2026-06-03T18:11:47Z","snapshot_observed_at":"2026-08-05T09:58:17.464593Z","submitted_at":"2025-10-27T16:32:12Z","title":"VOLD: Reasoning Transfer from LLMs to Vision-Language Models via On-Policy Distillation","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-04T07:57:25.863201Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2510.23497"},"observation_digest":"sha256:e13fc7d1cae42e40549173503015108297977f497ddbc514f4f21307f6505f04","observation_id":"853481f2-d296-431e-9a06-221d526c880a","resolution":{"observed_at":"2026-08-04T07:57:25.863201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2510.25741","last_updated":"2026-07-01T23:25:58Z","snapshot_observed_at":"2026-08-04T07:30:51.188041Z","submitted_at":"2025-10-29T17:45:42Z","title":"Scaling Latent Reasoning via Looped Language Models","version":4},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-15T07:43:11.620446Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2510.25741"},"observation_digest":"sha256:a500f24603ece55e5f5946d5b06ea01be979b83837c2915dfb38f65185689b6f","observation_id":"a3d3ceef-5195-4d7c-b4ee-8e0cb89b9aca","resolution":{"observed_at":"2026-05-15T07:43:11.776534Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-04T07:31:49.282466Z","title":"MMLU-Pro: A more challenging and reliable evaluation for massive multitask language understanding.arXiv preprintarXiv:2406.01574, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.25741","last_updated":"2026-07-01T23:25:58Z","snapshot_observed_at":"2026-08-04T07:30:51.188041Z","submitted_at":"2025-10-29T17:45:42Z","title":"Scaling Latent Reasoning via Looped Language Models","version":5},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-04T07:31:49.282466Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2510.25741"},"observation_digest":"sha256:7a3ad73657524ce95627100f43246febb82b8e805009e338fe5cbfb47bbea18a","observation_id":"b4bca95f-70eb-4909-bf5a-99a090fc6780","resolution":{"observed_at":"2026-08-04T07:31:49.282466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2511.07885","last_updated":"2026-05-21T03:40:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-11T06:33:30Z","title":"Intelligence per Watt: Measuring Intelligence Efficiency of Local AI","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2511.07885"},"observation_digest":"sha256:c25e4d722436c2cea06da0a591e82317e65edd2f0fe702dff9a3599357a81cee","observation_id":"9e592b64-9b71-43cf-a18a-f13e60f0fc48","resolution":{"observed_at":"2026-05-22T12:21:30.884331Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2511.15408","last_updated":"2026-05-14T06:10:01Z","snapshot_observed_at":"2026-07-06T22:36:24.466921Z","submitted_at":"2025-11-19T13:05:25Z","title":"Chinese Short-Form Creative Content Generation via Explanation-Oriented Multi-Objective Optimization","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:53.673037Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2511.15408"},"observation_digest":"sha256:cb8e6414c46f40c9c6f5d271619fd3febb1af4c593ea7802c6ca70024cc9cf5b","observation_id":"2d48a48f-0626-4c04-b14f-446aaa904da4","resolution":{"observed_at":"2026-05-17T20:52:06.100355Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2511.20857","last_updated":"2026-05-18T16:18:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-25T21:08:07Z","title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-14T23:13:15.016486Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2511.20857"},"observation_digest":"sha256:4e5a0ac0092b14ab45347d53e35613d84160484aedcfdac717e6e92669d60ca1","observation_id":"a4e6419c-84b2-4293-a891-e7e7ae4d1edf","resolution":{"observed_at":"2026-05-14T23:13:15.540596Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2511.21686","last_updated":"2026-04-18T01:35:07Z","snapshot_observed_at":"2026-08-01T03:26:40.746820Z","submitted_at":"2025-11-26T18:59:28Z","title":"Matrix: Peer-to-Peer Multi-Agent Synthetic Data Generation Framework","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T04:23:03.393566Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2511.21686"},"observation_digest":"sha256:aad292272916784fde7b59bf180dcb5f612eaf9afd64c65b63769f87c09b16bb","observation_id":"e6b570cf-1747-483d-867b-ea5c69fabbf1","resolution":{"observed_at":"2026-05-17T04:24:00.338380Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-07-31T23:49:25.878472Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T13:05:26.667750Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2512.02556"},"observation_digest":"sha256:414668f17da60082a06dc1611aefe201a8c3bb3df39e912752381c0bc7b2c7cf","observation_id":"ca3ea0aa-fbef-42cf-a952-6f954ac057f4","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-07-31T23:49:25.878472Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T13:05:26.667750Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2512.02556"},"observation_digest":"sha256:963417d118e6e9f00c0ae692b930b52c8e65e651afad8a00a221f89563a23a5a","observation_id":"d56cf372-e911-493c-b755-c56914d404c9","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-03T16:43:57.348223Z","title":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E., Le, Q., and Zhou, D","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.12576","last_updated":"2026-05-23T02:48:11Z","snapshot_observed_at":"2026-08-03T16:43:53.189781Z","submitted_at":"2025-12-14T07:03:51Z","title":"Coupled Variational Reinforcement Learning for Language Model General Reasoning","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T16:43:57.348223Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2512.12576"},"observation_digest":"sha256:1fb57dcab7176f3eb0fe5c656c54c6e97c09d3a0ccde010dbb71acf9b34d98d0","observation_id":"1bb6e91e-f99c-434f-94c0-21869bc5e718","resolution":{"observed_at":"2026-08-03T16:43:57.348223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2512.20856","last_updated":"2025-12-24T00:24:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-24T00:24:05Z","title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","version":1},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-05-18T01:40:42.190369Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2512.20856"},"observation_digest":"sha256:88b4338102612dda36746995b68682dd312819f72b11ef7a9190aaae5dacaa3f","observation_id":"37935a84-316a-48d8-b77d-f9260d32168a","resolution":{"observed_at":"2026-05-18T01:40:42.626283Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2601.04809","last_updated":"2026-05-04T13:55:24Z","snapshot_observed_at":"2026-07-06T22:41:05.793337Z","submitted_at":"2026-01-08T10:42:04Z","title":"SCALER:Synthetic Scalable Adaptive Learning Environment for Reasoning","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T16:24:04.132572Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2601.04809"},"observation_digest":"sha256:c37abc8e3fecdb63a7aaf1c0946b731ec5c261962bb92af365f9a272fec11186","observation_id":"501a36c9-b6a6-42da-83c3-236f8a557516","resolution":{"observed_at":"2026-05-16T16:28:05.717152Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-03T05:28:15.282398Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.02244","last_updated":"2026-07-14T10:18:39Z","snapshot_observed_at":"2026-08-06T11:19:20.172065Z","submitted_at":"2026-02-02T15:53:55Z","title":"Entropy-Preserving Supervised Fine-Tuning via Adaptive Self-Distillation for Large Reasoning Models","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-03T05:28:15.282398Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.02244"},"observation_digest":"sha256:4d71810718d461bd6a6c52550542d9db887319bd1c2cc51879b005b00c562e06","observation_id":"a43d7341-ecf4-40a2-9701-f2a8a24dab06","resolution":{"observed_at":"2026-08-03T05:28:15.282398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-10T16:09:05.225767Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.02276"},"observation_digest":"sha256:5892a30b77c913cdfdbcef321b4b9eb764bf1b8109e0bf2903fba5c5501b8369","observation_id":"afa417c7-b8ee-4eeb-99ee-f515404fff38","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-03T04:48:03.446902Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.04101","last_updated":"2026-06-02T20:12:27Z","snapshot_observed_at":"2026-08-03T23:55:28.063760Z","submitted_at":"2026-02-04T00:36:37Z","title":"Interfaze: The Future of AI is built on Task-Specific Small Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T04:48:03.446902Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.04101"},"observation_digest":"sha256:0367c9c6b5229105926fc7e9dca42fcf9c611b71a36235af3690ebd394dc1860","observation_id":"44893d04-a0d4-4ebf-91d0-affb247be254","resolution":{"observed_at":"2026-08-03T04:48:03.446902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2602.05946","last_updated":"2026-05-11T01:44:43Z","snapshot_observed_at":"2026-08-02T10:13:32.692710Z","submitted_at":"2026-02-05T18:01:52Z","title":"f-GRPO and Beyond: Divergence-Based Reinforcement Learning Algorithms for General LLM Alignment","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T06:48:35.723474Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.05946"},"observation_digest":"sha256:546aa30fa5e271a7540c5e43246ce63cbdd100b97cd15ae4a79fefaeb2104f01","observation_id":"788abe45-388b-43ba-bbf4-97259f34881d","resolution":{"observed_at":"2026-05-16T06:50:42.985357Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-03T02:51:56.936714Z","title":"Mmlu-pro: A more robust and challenging multi- task language understanding benchmark, 2024b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.09689","last_updated":"2026-06-18T12:30:19Z","snapshot_observed_at":"2026-08-06T07:48:42.751296Z","submitted_at":"2026-02-10T11:44:19Z","title":"Model soups need only one ingredient","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T02:51:56.936714Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.09689"},"observation_digest":"sha256:870859fe50045b42f6f70c928668195f1750c10d37a19ef7730bc69e9faba87c","observation_id":"b1f77faf-3a78-4387-a714-1653c3f5934d","resolution":{"observed_at":"2026-08-03T02:51:56.936714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2602.10144","last_updated":"2026-05-06T17:38:45Z","snapshot_observed_at":"2026-08-03T13:18:35.790856Z","submitted_at":"2026-02-09T10:45:13Z","title":"When LLMs get significantly worse: A statistical approach to detect model degradations","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T06:03:37.788377Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.10144"},"observation_digest":"sha256:23971ed5f4b3cdb425447b3f09e11757f2b74f696effdb126777252669c1ad59","observation_id":"c3d75179-36ee-422b-a25e-1390f51e1c74","resolution":{"observed_at":"2026-05-16T06:07:25.787095Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2602.10718","last_updated":"2026-04-28T04:57:29Z","snapshot_observed_at":"2026-07-06T22:45:25.047172Z","submitted_at":"2026-02-11T10:24:42Z","title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T05:58:03.113220Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.10718"},"observation_digest":"sha256:f2921ff63957246c41a3d0c0bebb0716155f34f38bcdab6853fe2539fcb1cce7","observation_id":"ab9cbd9a-41aa-410b-9fb2-8beda59b1d44","resolution":{"observed_at":"2026-05-16T06:00:40.744485Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-02T22:58:14.583482Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark.arXiv preprint arXiv:2406.01574,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.15327","last_updated":"2026-06-06T22:17:18Z","snapshot_observed_at":"2026-08-02T22:58:08.817247Z","submitted_at":"2026-02-17T03:13:51Z","title":"Prescriptive Scaling Reveals the Evolution of Language Model Capabilities","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T22:58:14.583482Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.15327"},"observation_digest":"sha256:37d592b6937bfc55684db7c971d51760909aef3a4a9668564ce930b2dfc8f5f9","observation_id":"b72c55e7-121f-469c-a4df-735975774b2a","resolution":{"observed_at":"2026-08-02T22:58:14.583482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-02T22:31:50.154870Z","title":"findings-acl.349","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.16763","last_updated":"2026-06-29T17:01:58Z","snapshot_observed_at":"2026-08-02T22:31:47.931184Z","submitted_at":"2026-02-18T16:51:37Z","title":"When AI Benchmarks Plateau: A Systematic Study of Benchmark Saturation","version":3},"reference_index":349,"source":"pdf_text","source_observed_at":"2026-08-02T22:31:50.154870Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.16763"},"observation_digest":"sha256:c278df31ae4d45c31af046548fb2fab10730a9dd36e8b5c454c1e49f6ca23a51","observation_id":"caeb7d52-bc16-4fac-a8f0-5664799f36bc","resolution":{"observed_at":"2026-08-02T22:31:50.154870Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-02T21:05:26.265009Z","title":"Mmlu-pro: A more robust and challenging multi-task language under- standing benchmark.arXiv preprint arXiv:2406.01574,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.21492","last_updated":"2026-07-18T04:56:29Z","snapshot_observed_at":"2026-08-04T06:06:14.689327Z","submitted_at":"2026-02-25T01:54:50Z","title":"GradAlign: Gradient-Aligned Data Selection for LLM Reinforcement Learning","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T21:05:26.265009Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2602.21492"},"observation_digest":"sha256:b86ba5722e4e0f34409216ff4656bafd6d6ac394fb7a9d12ceba37ce20acc46a","observation_id":"f43f4cb7-6c5a-45bc-a725-120c585397cc","resolution":{"observed_at":"2026-08-02T21:05:26.265009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2603.06610","last_updated":"2026-05-22T08:27:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-19T09:46:24Z","title":"CapTrack: Multifaceted Evaluation of Forgetting in LLM Post-Training","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-25T06:40:51.046965Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2603.06610"},"observation_digest":"sha256:49736bf951831c4aa6bc38d4edea3a5d081da3d4726263f7458162c1024512bf","observation_id":"2f83f8fb-db11-4b5a-b2bb-7a190450b54c","resolution":{"observed_at":"2026-05-25T06:45:26.414366Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-02T17:39:43.379650Z","title":"Mmlu-pro: A more ro- bust and challenging multi-task language understanding benchmark.arXiv preprint arXiv:2406.01574,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.23171","last_updated":"2026-07-29T15:06:33Z","snapshot_observed_at":"2026-08-06T12:57:45.987647Z","submitted_at":"2026-03-24T13:13:23Z","title":"Adaptively Robust LLM Monitoring via Activation Watermarking","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T17:39:43.379650Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2603.23171"},"observation_digest":"sha256:7bd87512f71a9224a469d7eedda9a6f22204cbb6d60fce061548247d69875704","observation_id":"80efb0bd-9825-4256-92e4-a6d350c0c80c","resolution":{"observed_at":"2026-08-02T17:39:43.379650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.02795","last_updated":"2026-04-03T07:02:57Z","snapshot_observed_at":"2026-07-06T22:52:06.460027Z","submitted_at":"2026-04-03T07:02:57Z","title":"Rubrics to Tokens: Bridging Response-level Rubrics and Token-level Rewards in Instruction Following Tasks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T20:24:17.788697Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.02795"},"observation_digest":"sha256:b09b9323d201a3d76be1e598c5905c0b5c6b1cf63d7f287f13f184fb9cc61ec7","observation_id":"6a2f23db-da70-419b-984e-dde9af5f0f48","resolution":{"observed_at":"2026-05-13T20:28:14.085645Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.03742","last_updated":"2026-04-04T14:07:37Z","snapshot_observed_at":"2026-08-04T22:20:40.535525Z","submitted_at":"2026-04-04T14:07:37Z","title":"Structured Multi-Criteria Evaluation of Large Language Models with Fuzzy Analytic Hierarchy Process and DualJudge","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T17:12:07.852282Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.03742"},"observation_digest":"sha256:ff501620c3b3749d439607ee36ad724d755bed5e4f6f33ba04f056f1a657b540","observation_id":"6c763adf-ed5f-4394-9af3-ae316fca3616","resolution":{"observed_at":"2026-05-13T17:13:01.112191Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.03993","last_updated":"2026-04-05T06:30:50Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T06:30:50Z","title":"Can LLMs Learn to Reason Robustly under Noisy Supervision?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T16:58:42.129870Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.03993"},"observation_digest":"sha256:4d07633a9fa1b44ab3e83601f23fb015d7468be65c7cbd5e30017214c8040df1","observation_id":"d78aedea-1cda-495f-b03e-bc2ae3a7bcbd","resolution":{"observed_at":"2026-05-13T17:08:01.249186Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.06628","last_updated":"2026-04-08T03:11:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-08T03:11:16Z","title":"Rethinking Generalization in Reasoning SFT: A Conditional Analysis on Optimization, Data, and Model Capability","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T18:53:22.553430Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.06628"},"observation_digest":"sha256:977be1103fb87e9c30a31bd69aadadfbb6459be24744291dfc5d234e0c409217","observation_id":"a1659414-7bad-4690-9125-b7331fa6d383","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-07-13T08:42:28.957233Z","title":"Chengyue Wu, Hao Zhang, Shuchen Xue, Zhijian Liu, Shizhe Diao, Ligeng Zhu, Ping Luo, Song Han, and Enze Xie","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.07022","last_updated":"2026-06-10T20:32:46Z","snapshot_observed_at":"2026-08-02T08:46:22.062564Z","submitted_at":"2026-04-08T12:40:09Z","title":"An Algebraic Introduction to Persistence","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-13T08:42:28.957233Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.07022"},"observation_digest":"sha256:93b8c52acb10c522f1311c1b7348ec22d8b5465568beb09f848d24d4e3160ed9","observation_id":"0ca98958-f293-46bc-aeb3-13faeff71956","resolution":{"observed_at":"2026-07-13T08:42:28.957233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.07023","last_updated":"2026-04-08T12:41:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-08T12:41:37Z","title":"MARS: Enabling Autoregressive Models Multi-Token Generation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T17:54:16.289777Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.07023"},"observation_digest":"sha256:499f8c5b0cafb9492925b77c23942dbe4ad10e402f4895328a3560df63854229","observation_id":"4e138d32-793c-4c50-bbc2-89e73ec0668f","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.08644","last_updated":"2026-04-09T17:51:11Z","snapshot_observed_at":"2026-08-02T14:49:38.234793Z","submitted_at":"2026-04-09T17:51:11Z","title":"EXAONE 4.5 Technical Report","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T17:47:34.692414Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.08644"},"observation_digest":"sha256:a64438817ab13af776ffa25274506bf3dfc747e92bfd05e2bd79b3b235ac6ea3","observation_id":"1d9c6c63-4821-4781-943b-05804e029e3b","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.09813","last_updated":"2026-04-10T18:38:52Z","snapshot_observed_at":"2026-07-06T22:58:34.504325Z","submitted_at":"2026-04-10T18:38:52Z","title":"Controllable and Verifiable Tool-Use Data Synthesis for Agentic Reinforcement Learning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T17:42:57.596073Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.09813"},"observation_digest":"sha256:2cda602b17f2c2064404e1c0878373a284e13b3c23a0e9f36caeb777a9dc70f3","observation_id":"79092f1e-a2f9-4337-bc51-0762e73a6eab","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.14168","last_updated":"2026-03-24T09:03:16Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-03-24T09:03:16Z","title":"SAGE Celer 2.6 Technical Card","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T00:54:24.133637Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.14168"},"observation_digest":"sha256:e06bf399bac2a69eaa49831be3242ec938f87a287ba2b236b068adc384f56f6d","observation_id":"4dfae951-1f02-4ee3-9604-a503b70e2e7a","resolution":{"observed_at":"2026-05-15T00:59:36.852198Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.18239","last_updated":"2026-07-23T15:30:50Z","snapshot_observed_at":"2026-08-02T15:57:09.270650Z","submitted_at":"2026-04-20T13:23:27Z","title":"Towards Disentangled Preference Optimization Dynamics: Suppress the Loser, Preserve the Winner","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T05:00:10.325345Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.18239"},"observation_digest":"sha256:d83a6623221ca2b32b91fbc15e81f107040c6c7f9e6148da7bd6c2218c68e739","observation_id":"bf083123-c623-418c-9b18-41476d5f7bac","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-02T15:57:22.150586Z","title":"org/CorpusID:46615544","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2604.18239","last_updated":"2026-07-23T15:30:50Z","snapshot_observed_at":"2026-08-02T15:57:09.270650Z","submitted_at":"2026-04-20T13:23:27Z","title":"Towards Disentangled Preference Optimization Dynamics: Suppress the Loser, Preserve the Winner","version":4},"reference_index":2012,"source":"pdf_text","source_observed_at":"2026-08-02T15:57:22.150586Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.18239"},"observation_digest":"sha256:3685984d9926afa419cc33e898dfe8e39167ade2cd9befa0193bd7853c286f61","observation_id":"c3b268b7-6730-49c6-b7a8-8003c6578aac","resolution":{"observed_at":"2026-08-02T15:57:22.150586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.18738","last_updated":"2026-05-07T10:54:12Z","snapshot_observed_at":"2026-08-02T09:44:06.500717Z","submitted_at":"2026-04-20T18:43:28Z","title":"Remask, Don't Replace: Token-to-Mask Refinement in Diffusion Large Language Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T04:57:48.890310Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.18738"},"observation_digest":"sha256:202e7bfd40a601a9d2d816a85f90ccde935879367b2eb3c23ac5a5b4d9544c3a","observation_id":"ce3e64bc-bbab-4053-b9e2-9ba504ee56bf","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.19877","last_updated":"2026-04-21T18:00:25Z","snapshot_observed_at":"2026-07-06T23:06:26.596990Z","submitted_at":"2026-04-21T18:00:25Z","title":"Super Apriel: One Checkpoint, Many Speeds","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-10T02:27:11.553553Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.19877"},"observation_digest":"sha256:d335353e87ec5e1d1e93d54fcadc4e794837cfbcc1067955172307d67423cd5d","observation_id":"f0e7dc22-40c6-45a3-be3e-5f64771b33ab","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.20500","last_updated":"2026-04-22T12:42:03Z","snapshot_observed_at":"2026-08-03T00:39:38.077432Z","submitted_at":"2026-04-22T12:42:03Z","title":"Efficient Test-Time Inference via Deterministic Exploration of Truncated Decoding Trees","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T01:03:06.917872Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.20500"},"observation_digest":"sha256:c7c6e2b279f8e14a11cfeff5d6c81ca76fc8216a406ff761d9c57f076a8e51cd","observation_id":"c3b5426d-f612-473c-83dc-df844eb8b772","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.20658","last_updated":"2026-04-22T15:07:54Z","snapshot_observed_at":"2026-08-02T11:39:51.117378Z","submitted_at":"2026-04-22T15:07:54Z","title":"Cooperative Profiles Predict Multi-Agent LLM Team Performance in AI for Science Workflows","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-09T23:52:47.833809Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.20658"},"observation_digest":"sha256:5f152a9ab55cdbdd882eaddfafd7228b43bf5d940da81da949ceaf2cbaf451cd","observation_id":"400845a3-79bc-45ec-9893-386c1ea14dc4","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.20720","last_updated":"2026-04-22T16:07:10Z","snapshot_observed_at":"2026-07-06T23:07:25.185196Z","submitted_at":"2026-04-22T16:07:10Z","title":"COMPASS: COntinual Multilingual PEFT with Adaptive Semantic Sampling","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-10T01:14:16.831333Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.20720"},"observation_digest":"sha256:74c60cbdc89b627c2d10baefef90917292ce179df83ba6ba55374174b14419d5","observation_id":"2cb20aaf-5490-4d74-aab3-b8f6c07abe71","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.21428","last_updated":"2026-04-23T08:45:38Z","snapshot_observed_at":"2026-07-06T23:08:01.670049Z","submitted_at":"2026-04-23T08:45:38Z","title":"Decoupled DiLoCo for Resilient Distributed Pre-training","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-09T22:20:21.090246Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.21428"},"observation_digest":"sha256:c56cf102b3b8b747f975ad1678f9e7988243a65579e60d0e36be63e62ddd271e","observation_id":"45f9dfce-013a-4c58-8ccf-878cbf61be79","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.22266","last_updated":"2026-04-24T06:26:24Z","snapshot_observed_at":"2026-07-06T23:08:42.301487Z","submitted_at":"2026-04-24T06:26:24Z","title":"Large Language Models Decide Early and Explain Later","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-08T12:06:33.200365Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.22266"},"observation_digest":"sha256:7d7c684f0d9361568384b4f00516a54623de2100b14acee069244574ba771ddf","observation_id":"2cfb4e89-8f68-4fcc-bf3b-8e3619686296","resolution":{"observed_at":"2026-05-11T19:21:09.335968Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.26206","last_updated":"2026-04-29T01:23:34Z","snapshot_observed_at":"2026-07-06T23:11:52.954360Z","submitted_at":"2026-04-29T01:23:34Z","title":"Option-Order Randomisation Reveals a Distributional Position Attractor in Prompted Sandbagging","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-07T13:38:37.045980Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.26206"},"observation_digest":"sha256:5f430ecc7d81c3566210d86e3bcc7faebcfd7b749532a934a819cb3c50ae7d7c","observation_id":"6418bdca-2c56-4701-8d70-70b495f8ae2b","resolution":{"observed_at":"2026-05-12T08:51:24.455720Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2604.27249","last_updated":"2026-04-29T22:48:24Z","snapshot_observed_at":"2026-07-06T23:12:43.284564Z","submitted_at":"2026-04-29T22:48:24Z","title":"Instruction Complexity Induces Positional Collapse in Adversarial LLM Evaluation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-07T09:08:04.485654Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2604.27249"},"observation_digest":"sha256:019f9ffadcb05a2a7c8b6b82b46f84245ca800fa2f8e60ed90443bfa90ed8fdc","observation_id":"0f200792-a149-4cb1-9cdc-88d7d63c1b3a","resolution":{"observed_at":"2026-05-12T09:46:28.452415Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.02255","last_updated":"2026-05-04T06:06:41Z","snapshot_observed_at":"2026-08-01T02:40:07.053714Z","submitted_at":"2026-05-04T06:06:41Z","title":"On the Privacy of LLMs: An Ablation Study","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-08T18:25:05.586464Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.02255"},"observation_digest":"sha256:d64c2dbca950713a8a61184fdcb9e59f6cd7a07dc767957fdcec55d89dbb685f","observation_id":"e47e59a7-39f5-4d39-9311-06e1dc67e90f","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.02930","last_updated":"2026-04-27T18:07:58Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-27T18:07:58Z","title":"Analysis and Explainability of LLMs Via Evolutionary Methods","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-09T20:37:40.932811Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.02930"},"observation_digest":"sha256:7fad090642dfbe7beccc41691e67eaedb09f08f4e7ea8e88559d93468e0cdcfe","observation_id":"c3543f14-d254-456a-bc90-e59de6e7622f","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.05175","last_updated":"2026-05-06T17:42:01Z","snapshot_observed_at":"2026-07-06T23:17:48.161262Z","submitted_at":"2026-05-06T17:42:01Z","title":"MRI-Eval: A Tiered Benchmark for Evaluating LLM Performance on MRI Physics and GE Scanner Operations Knowledge","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-08T15:33:47.276997Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.05175"},"observation_digest":"sha256:be60a560c86f2b223d0f59af87a60271d66f40ce658e47cec9a7af83b3bd1294","observation_id":"7c459372-68c1-4500-ad77-d9fc3986ac3a","resolution":{"observed_at":"2026-05-11T18:36:06.891413Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.05810","last_updated":"2026-05-07T07:46:17Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T07:46:17Z","title":"CXR-ContraBench: Benchmarking Negated-Option Attraction in Medical VLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-08T14:49:53.357083Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.05810"},"observation_digest":"sha256:f86bfc94803f8b286dbbab089de8bc14442158d57e23953685be04ce9ff0063d","observation_id":"a1e9628e-5de0-4c35-82db-d6a5ed30e9a3","resolution":{"observed_at":"2026-05-11T18:41:09.522954Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.06387","last_updated":"2026-05-13T04:44:13Z","snapshot_observed_at":"2026-07-06T23:18:51.157048Z","submitted_at":"2026-05-07T15:02:49Z","title":"Asymmetric On-Policy Distillation: Bridging Exploitation and Imitation at the Token Level","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-08T12:57:24.822525Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.06387"},"observation_digest":"sha256:d0ad60c9a1ab51d5eea2ac34722ab99b99282975d397b5fcb9991a2f6b13b114","observation_id":"571dec5d-bd09-4e02-b987-0d54233a5938","resolution":{"observed_at":"2026-05-11T19:01:17.462018Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.06387","last_updated":"2026-05-13T04:44:13Z","snapshot_observed_at":"2026-07-06T23:18:51.157048Z","submitted_at":"2026-05-07T15:02:49Z","title":"Asymmetric On-Policy Distillation: Bridging Exploitation and Imitation at the Token Level","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-11T01:51:29.021423Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.06387"},"observation_digest":"sha256:7b6db9f60ee1fba3dbc23a2b6a34d11d5f9790d9906b874c82ccad74df6aabb8","observation_id":"f5f17757-15a0-4b7f-b6e5-ced88756b3a4","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.06387","last_updated":"2026-05-13T04:44:13Z","snapshot_observed_at":"2026-07-06T23:18:51.157048Z","submitted_at":"2026-05-07T15:02:49Z","title":"Asymmetric On-Policy Distillation: Bridging Exploitation and Imitation at the Token Level","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-14T21:14:22.718769Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.06387"},"observation_digest":"sha256:78c0c50c0fc36fcf396847e54b44478be95f0543e57dd4954c7e1d971151981e","observation_id":"7005586a-ccad-47d4-b341-0ef14cc2ac73","resolution":{"observed_at":"2026-05-14T21:17:59.726475Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.07046","last_updated":"2026-05-07T23:52:12Z","snapshot_observed_at":"2026-07-06T23:19:25.538514Z","submitted_at":"2026-05-07T23:52:12Z","title":"An Interpretable and Scalable Framework for Evaluating Large Language Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-11T01:08:25.577363Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.07046"},"observation_digest":"sha256:80c9b088857dcdaf7aed8b481d8d1eb7b968efa5c74398e1b0f1e3c316c4609f","observation_id":"2746932e-ad2b-475b-9cf0-99433ca2815b","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.07182","last_updated":"2026-05-08T03:20:36Z","snapshot_observed_at":"2026-07-06T23:19:35.885430Z","submitted_at":"2026-05-08T03:20:36Z","title":"Star Elastic: Many-in-One Reasoning LLMs with Efficient Budget Control","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T01:03:28.231935Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.07182"},"observation_digest":"sha256:c0b2a6dfabb601adf7524b4c1900528197666693687586267c65fdaa2cc67ed8","observation_id":"309a17ee-5789-4d99-bb5d-11047c569070","resolution":{"observed_at":"2026-05-11T15:51:14.588885Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.08738","last_updated":"2026-05-18T06:29:11Z","snapshot_observed_at":"2026-07-06T23:20:57.084438Z","submitted_at":"2026-05-09T06:50:35Z","title":"SlimQwen: Exploring the Pruning and Distillation in Large MoE Model Pre-training","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-12T03:34:10.370956Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.08738"},"observation_digest":"sha256:dae7c5f30e389b2fb31c77fa4f7d1a8242013cd7d0092d34884c746a69a52267","observation_id":"3ffc6f3c-aebb-48eb-8c30-b12165231b2a","resolution":{"observed_at":"2026-05-12T07:16:28.647789Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.08738","last_updated":"2026-05-18T06:29:11Z","snapshot_observed_at":"2026-07-06T23:20:57.084438Z","submitted_at":"2026-05-09T06:50:35Z","title":"SlimQwen: Exploring the Pruning and Distillation in Large MoE Model Pre-training","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-20T23:22:51.808346Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.08738"},"observation_digest":"sha256:3d002cc5226ada13753b1e8cf34276b96ee6af4a9d46ea4f0c8cd1c4374bea7f","observation_id":"bb546503-eb3a-4299-9ff5-7413e0c653f4","resolution":{"observed_at":"2026-05-20T23:23:51.131371Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.09542","last_updated":"2026-05-10T13:54:37Z","snapshot_observed_at":"2026-07-06T23:21:35.327066Z","submitted_at":"2026-05-10T13:54:37Z","title":"LLM-Guided Monte Carlo Tree Search over Knowledge Graphs: Composing Mechanistic Explanations for Drug-Disease Pairs","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-12T03:04:57.158157Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.09542"},"observation_digest":"sha256:353ea4e7f21a526bea4809fdc8f38d5b4048a269fadf4a60f44ea8b138caa233","observation_id":"253b5cf3-edda-4978-82e5-d2ae8b28a274","resolution":{"observed_at":"2026-05-12T03:06:18.277379Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.10973","last_updated":"2026-05-08T20:20:05Z","snapshot_observed_at":"2026-07-06T23:22:52.369277Z","submitted_at":"2026-05-08T20:20:05Z","title":"Rotation-Preserving Supervised Fine-Tuning","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-13T06:26:20.393476Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.10973"},"observation_digest":"sha256:368ebcc7bbd520a05144a0fec4c2d980b22ae7590fe85309a505f8f07a395ec7","observation_id":"1ad04c0f-798c-4c0c-8e84-b0354f2f0ee8","resolution":{"observed_at":"2026-05-13T06:27:24.449817Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.11663","last_updated":"2026-05-12T07:22:50Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T07:22:50Z","title":"Human-Grounded Multimodal Benchmark with 900K-Scale Aggregated Student Response Distributions from Japan's National Assessment of Academic Ability","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-13T01:12:44.476894Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.11663"},"observation_digest":"sha256:7452193cc66eec766fe12ea46dc9b90f652ee9a854f5d320c08bfb608357ded9","observation_id":"2aa65d82-511a-41e7-9f90-88520182bc1a","resolution":{"observed_at":"2026-05-13T01:17:02.718420Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.14164","last_updated":"2026-05-13T22:39:10Z","snapshot_observed_at":"2026-07-06T23:25:39.415637Z","submitted_at":"2026-05-13T22:39:10Z","title":"Unsteady Metrics and Benchmarking Cultures of AI Model Builders","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-15T04:54:26.888562Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.14164"},"observation_digest":"sha256:6d97eb7dbeb4f9e803946d95f07bd14f80a56fd502c72a1459a5c9144d69b574","observation_id":"73b0d0dd-1501-492f-ac0c-8a5e4ccaedbd","resolution":{"observed_at":"2026-05-15T04:55:01.582395Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.14322","last_updated":"2026-08-02T07:39:44Z","snapshot_observed_at":"2026-08-06T17:31:43.159244Z","submitted_at":"2026-05-14T03:34:25Z","title":"TeachArena: Are Language Agents Ready for Realistic Teaching Work?","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T10:27:32.588593Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.14322"},"observation_digest":"sha256:ac6cdaef4005cc64038e36a6bbde864c598ee1fdda1b85ea28eea08738cad698","observation_id":"96fa0261-503c-4882-9cc0-c9bbbce191d5","resolution":{"observed_at":"2026-05-22T10:31:25.829749Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-04T05:12:48.986173Z","title":"MMLU-Pro: A more robust and challenging multi-task language understanding benchmark.arXiv preprint arXiv:2406.01574,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.14322","last_updated":"2026-08-02T07:39:44Z","snapshot_observed_at":"2026-08-06T17:31:43.159244Z","submitted_at":"2026-05-14T03:34:25Z","title":"TeachArena: Are Language Agents Ready for Realistic Teaching Work?","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T05:12:48.986173Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.14322"},"observation_digest":"sha256:4a306a2714d20a4b3537c919ceb9974d969a19054377cdea96f8b39dfbec192e","observation_id":"bc718502-e1b7-4c4a-8984-acb4bd8be7db","resolution":{"observed_at":"2026-08-04T05:12:48.986173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":"2406.01574","doi":"10.1145/3711896.3737413","metadata_source":"pith","pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","venue":"cs.CL","work_id":"3c028052-035a-4c22-b80e-3046edb44adc","year":2024},"citing_paper":{"arxiv_id":"2605.17361","last_updated":"2026-07-09T07:36:31Z","snapshot_observed_at":"2026-07-12T23:17:31.709829Z","submitted_at":"2026-05-17T09:58:58Z","title":"MasFACT: Continual Multi-Agent Topology Learning via Geometry-Aware Posterior Transfer","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-20T14:35:46.376752Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2605.17361"},"observation_digest":"sha256:0c24fc340425c989fe1811a9043a87f2a45975eed17f280236ccc21c7be7c189","observation_id":"85c8b5fd-c397-4ecc-b0e4-63273f66d950","resolution":{"observed_at":"2026-05-20T14:38:21.603192Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2406.01574/citation-record","integrity":"/paper/2406.01574/integrity","json":"/paper/2406.01574/citation-record.json","paper":"/paper/2406.01574"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":"2404.14219","doi":"10.48550/arxiv.2404.14219","metadata_source":"pith","pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","venue":"cs.CL","work_id":"feef9556-a016-493c-abd2-0c97a23a7ebf","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:24aab07881ffce0e0cb06001d3dda714386c52d6034bb1211ac93ac7548ed828","observation_id":"8234b9d2-1c13-457f-aebe-7d9712606e39","resolution":{"observed_at":"2026-05-11T15:51:08.647897Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:f0845386efc124e96cc94ebaa8fe2ca3947d2002345469203285c1ea53ce8073","observation_id":"8ab3c5ce-f484-4c17-abef-3279fbbc686f","resolution":{"observed_at":"2026-05-11T15:51:06.112897Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-07-31T17:52:54.507082Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":"2402.01781","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"When benchmarks are targets: Revealing the sensitivity of large language model leaderboards","venue":null,"work_id":"e17be6ab-246f-48c4-b76e-49c58b63b133","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:2e181de53f2a2ef020104433e211960060dfaa60683c43a4dab63363105062ed","observation_id":"a93ec630-105a-4a4c-a428-285fb95e7f65","resolution":{"observed_at":"2026-05-11T15:51:06.285386Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10631","last_updated":"2024-03-15T19:14:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-16T17:54:07Z","title":"Llemma: An Open Language Model For Mathematics","version":3},"cited_work":{"arxiv_id":"2310.10631","doi":"10.48550/arxiv.2310.10631","metadata_source":"pith","pith_arxiv_id":"2310.10631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llemma: An Open Language Model For Mathematics","venue":"cs.CL","work_id":"c0820dec-60a1-4f0c-beda-4516eb8e89c9","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2310.10631","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:656234c4cc7137492cd9bb15b9883e58fe10a4c91c141144fef8345c23100983","observation_id":"e33005bd-4ea5-4b2a-8f8e-67a422c7979b","resolution":{"observed_at":"2026-05-19T08:17:47.048289Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-21T13:53:00.912171+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T13:53:00.912171+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:377096abfe168db6824580e7c9b9cb333d769d4c674877a1f6dd50016b11d541","observation_id":"8ad7b15b-7468-4c6e-87d2-0c0b3c72b908","resolution":{"observed_at":"2026-05-11T15:51:08.512353Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":"2212.08073","doi":"10.48550/arxiv.2212.08073","metadata_source":"pith","pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Constitutional AI: Harmlessness from AI Feedback","venue":"cs.CL","work_id":"faaaa4e0-2676-4fac-a0b4-99aef10d2095","year":2022},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:d5c1ffa5ffc082c2d6ddefe7329f8bebc0134cc9909b1409ebe0724509a20f4d","observation_id":"07b26437-c504-4e4c-b354-436e5de5ce8d","resolution":{"observed_at":"2026-05-11T15:51:09.594468Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:15.114855+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:15.114855+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models are few-shot learners","venue":null,"work_id":"9c06112a-442e-4378-8ea7-f1e22339c0a9","year":1901},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:2b240d7ce785319d29d2ece1cae179116f4779d17f37bc82f0c12039212e882d","observation_id":"dae13852-ec92-4270-b9e0-c5c9fd915394","resolution":{"observed_at":"2026-05-11T15:51:10.387353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"C4ai-command-r-v01","venue":null,"work_id":"ce407bd6-2a6e-46e9-bb66-4f9595c8002f","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:17a36c088dd59a125e8d5c14a3e12526e5ba2c11363a65c2b25246061947988b","observation_id":"c5b68c9b-5794-42fe-a950-33507f13d739","resolution":{"observed_at":"2026-05-11T15:51:10.586258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A survey on evaluation of large language models","venue":null,"work_id":"4a3081a3-8d31-491a-a363-1f5bb42a5b24","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:0e596020118837c7958e12c387c888f17f707092db800c097417e1bff6c02c62","observation_id":"388eeb6b-0750-4eaa-bcb5-f5df8ef21cd4","resolution":{"observed_at":"2026-05-11T15:51:10.724417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Theoremqa: A theorem-driven question answering dataset","venue":null,"work_id":"6ead68f8-9b38-4863-9a03-dde38ce3113c","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:7dec381a282fc34bb9b57ff038c90216ceb7c852e81dceea6f2346a9f5999956","observation_id":"3c29c35f-b05c-4344-904a-b9625bb5bc31","resolution":{"observed_at":"2026-05-11T15:51:10.934346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:911a573637f921fd969c1796c13825a10d5b3d1c51df4ac9075d0f49b7e33014","observation_id":"8a07d66b-ea87-4efa-9588-0d8643c0835f","resolution":{"observed_at":"2026-05-13T15:13:25.541930Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":"1803.05457","doi":"10.1162/tacl_a_00448.https://aclanthology.org/2022.tacl-1.5","metadata_source":"pith","pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","venue":"cs.AI","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","year":2018},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:904c91c15246ed2f6415bf075d2bff97e6988ea158106c74387b6851700c933e","observation_id":"492b83d4-6c0d-4287-b9c8-94e1c1b41563","resolution":{"observed_at":"2026-05-11T15:51:09.355009Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing the next generation of Claude https://www.anthropic.com/news/claude-3- family","venue":null,"work_id":"f8619816-2c8d-448f-8152-be188caf85c5","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:3ae065278a2ef07a7e7379f962e5c51df82795393276b00c078cd9e78840b426","observation_id":"d76b1645-9be5-4808-a115-c7e196f71b2f","resolution":{"observed_at":"2026-05-11T15:51:11.065802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Opencompass: A universal evaluation platform for foundation models","venue":null,"work_id":"83572c5c-d5f1-4ee5-b6c0-93bd274e7c0c","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:83d78ba316da2296e02a268edaed29fc63e656f2490ca759df472fc626b88820","observation_id":"55ef595b-430f-4c61-b692-a5d75feb2c78","resolution":{"observed_at":"2026-05-11T15:51:11.144240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T11:54:51.813738Z","title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model","venue":null,"work_id":"54678429-f341-4795-b670-b852f6eabe23","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:ba4f4505698a750b8b626c60037ca42947bb1e8370afb4abe502a11f3d24c687","observation_id":"57632eba-9ea4-41e9-a3b7-291c12954239","resolution":{"observed_at":"2026-05-11T15:51:11.304857Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17306","last_updated":"2023-05-26T23:46:42Z","snapshot_observed_at":"2026-07-06T15:34:13.017210Z","submitted_at":"2023-05-26T23:46:42Z","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","version":1},"cited_work":{"arxiv_id":"2305.17306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.17306","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2305.17306 , year=","venue":null,"work_id":"bbda8612-8e45-46f9-96ea-1000603eaf6f","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2305.17306","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:8709c5abfa86bbaf3ea4c9367409be15d259b6908e3c1dc578b48e74e102e12f","observation_id":"fc57fbf5-dd0d-4f2f-b7e5-f2439d25b8ec","resolution":{"observed_at":"2026-05-11T15:51:06.479417Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hello gpt4-o","venue":null,"work_id":"fd80ff94-0ea4-4df9-b5a3-e7f252067970","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:bd2a35a9dda9684f292280e7ba61d52fe4ea6d1c544557ac957639c039f6b8dd","observation_id":"08d07f93-b368-442c-9580-e75518cd6ca7","resolution":{"observed_at":"2026-05-11T15:51:11.564521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":"2009.03300","doi":"10.48550/arxiv.2009.03300","metadata_source":"pith","pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Massive Multitask Language Understanding","venue":"cs.CY","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","year":2020},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:a48d20bea5d1f54c7f8d4d202e3dba6f836950d001366c7ddac4de27899c66d9","observation_id":"138b782d-5d60-4b69-8591-73fd4c777ae2","resolution":{"observed_at":"2026-05-11T15:51:07.557358Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:72cab7a785e3f1545e6c86acb228b0749b029c81b0e71a8d49554de21ff5dfe9","observation_id":"e186c4b4-df80-4dc1-b526-5b50baea891d","resolution":{"observed_at":"2026-05-11T15:51:07.937927Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":"2310.06825","doi":"10.48550/arxiv.2310.06825","metadata_source":"pith","pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mistral 7B","venue":"cs.CL","work_id":"eb5e1305-ad11-4875-ad8d-ad8b8f697599","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:4c05f050f9c023eb8e8ee4d40bb8faffae5e341b5cfa8a36014322737494855c","observation_id":"a66e1811-fd49-41e9-b8fd-45cfc6627eb9","resolution":{"observed_at":"2026-05-11T15:51:08.055161Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:12.282824+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:12.282824+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":"2401.04088","doi":"10.48550/arxiv.2401.04088","metadata_source":"pith","pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mixtral of Experts","venue":"cs.LG","work_id":"0de8c352-9daa-4e1e-8c7b-3d0dec69f369","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:093543d75350349f7958f2a0ae2dfc492a655ada562d015359ad818edf8019d0","observation_id":"cc3eb58a-ed94-42c8-a088-2ca4bbe6a028","resolution":{"observed_at":"2026-05-11T15:51:08.257362Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-09T08:48:39.110013+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T08:48:39.110013+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":"2211.09110","doi":"10.1007/bf01194075","metadata_source":"pith","pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Holistic Evaluation of Language Models","venue":"cs.CL","work_id":"cc02a01e-7218-47dc-8e66-3333e7e4adec","year":2022},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:d471c1c838a883d1c382d14a40a728abe571727bddef983547eb0ef2c7564939","observation_id":"8a3635d8-5b20-4c17-9a12-330413744d0b","resolution":{"observed_at":"2026-05-11T15:51:08.467354Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lingyiwanwu, yi-large","venue":null,"work_id":"128e959a-50ed-4b3e-be66-6941afdfdeb9","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:0ee1013c930642ef4a410addf36493253729188ef7283681f05a168443ec92fe","observation_id":"6f640b04-8a51-4717-944e-9497af619c71","resolution":{"observed_at":"2026-05-11T15:51:11.788363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Build the future of ai with meta llama 3 - https://llama.meta.com/llama3/","venue":null,"work_id":"d4b4169a-64d1-45cd-affb-974762917aa5","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:700eea667f43f5a1c9c1ae7cdbaeb07640ce2f3845421f3e2483ed92a08d8d12","observation_id":"6f2251c1-4efa-4ced-a393-b3b2eba62bcc","resolution":{"observed_at":"2026-05-11T15:51:12.017351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.08773","last_updated":"2022-03-14T09:15:08Z","snapshot_observed_at":"2026-08-05T18:13:22.401467Z","submitted_at":"2021-04-18T08:44:56Z","title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","version":4},"cited_work":{"arxiv_id":"2104.08773","doi":"10.1002/wps.20513","metadata_source":"pith","pith_arxiv_id":"2104.08773","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","venue":"cs.CL","work_id":"807e5cef-0bb3-4eae-a684-154fc2d6aa06","year":2021},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2104.08773","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:798b9a5b7ae7fd96fe2cff366bd7444c4b31222caecfe70f862d71096d3027ab","observation_id":"ea39cfd5-e27a-4d09-ba65-aaca128c59da","resolution":{"observed_at":"2026-05-18T01:57:29.846007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2311.02462","doi":"10.48550/arxiv.2311.02462","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Levels of agi: Opera- tionalizing progress on the path to agi","venue":"arXiv (Cornell University)","work_id":"befdfddb-714c-423e-a5ab-85dd7ff5f0c5","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:09bb7d8609ac150b02cc7a0127fb6815282532a7dfb46b1e4f422f31a472370c","observation_id":"14d1b2fb-8384-495c-afbc-8b036362b99c","resolution":{"observed_at":"2026-05-11T15:51:09.094372Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open LLM Leaderboard - a Hugging Face Space by open-llm-leaderboard","venue":null,"work_id":"d093882e-46f5-4a44-ba25-eab061d0c56a","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:857d307222804738344cec6aca5d215087785b862c01aa2a0cf379ca50000f82","observation_id":"45418f88-7c29-4464-a544-f49336cdac0f","resolution":{"observed_at":"2026-05-11T15:51:12.336355Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"a0c5696a-8445-461a-9639-21b8667d28f5","year":2022},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:da012b4b45f132053d9841c5dcae7046776081f2ea96a68cfbd72e5ed48e68f8","observation_id":"891b46ca-3b5b-424d-b55f-257ac0ee9273","resolution":{"observed_at":"2026-05-11T15:51:12.858846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13228","last_updated":"2024-07-03T13:46:33Z","snapshot_observed_at":"2026-07-31T05:05:41.080329Z","submitted_at":"2024-02-20T18:42:34Z","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","version":2},"cited_work":{"arxiv_id":"2402.13228","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.13228","snapshot_observed_at":"2026-07-03T16:48:40.350225Z","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","venue":"cs.CL","work_id":"b220de8b-d5cf-4eef-9841-1428c753012c","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2402.13228","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:46e1aaae0716f19ff8186bac23459fbb9478103c0b9f3aef51785df7187f0604","observation_id":"c7a9afcc-b632-4e4e-8d02-8dd740516664","resolution":{"observed_at":"2026-05-17T23:04:44.937279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:f306647bbd137d6beacc4b1a9c95a8786102aa7ba9868ae080c2c04adcd1b99e","observation_id":"d6e4ec4d-23d5-4def-9ea6-9d2dfe73c24e","resolution":{"observed_at":"2026-05-11T15:51:09.784439Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.08207","last_updated":"2022-03-17T17:53:01Z","snapshot_observed_at":"2026-07-06T11:58:21.596920Z","submitted_at":"2021-10-15T17:08:57Z","title":"Multitask Prompted Training Enables Zero-Shot Task Generalization","version":3},"cited_work":{"arxiv_id":"2110.08207","doi":"10.48550/arxiv.2110.08207","metadata_source":"pith","pith_arxiv_id":"2110.08207","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Multitask Prompted Training Enables Zero-Shot Task Generalization","venue":"cs.LG","work_id":"1857c952-db2f-4810-9e3c-27a862c8d276","year":2021},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2110.08207","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:e9df850b47bb52aa3eee2d30caa3c6d1d1046bedbed797a945653445fc77f398","observation_id":"215a3e30-542b-4e41-855b-8aa9356a55d2","resolution":{"observed_at":"2026-05-14T17:59:43.395267Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":"2206.04615","doi":"10.1162/tacl_a_00688","metadata_source":"pith","pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","venue":"cs.CL","work_id":"bb63abb3-0d50-4362-b97c-b5e725b03b39","year":2022},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:de2190cf1bc073cdc9d521e1c348f7b299d78b6c3b8f9f9bd282611a53bb870f","observation_id":"6dc02a4e-23d8-4905-8760-e3009390288a","resolution":{"observed_at":"2026-05-11T15:51:05.108235Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09261","last_updated":"2022-10-17T17:08:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-17T17:08:26Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","version":1},"cited_work":{"arxiv_id":"2210.09261","doi":"10.48550/arxiv.2210.09261","metadata_source":"pith","pith_arxiv_id":"2210.09261","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","venue":"cs.CL","work_id":"513eb205-04ca-4722-9a43-a74e8cbe7e85","year":2022},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2210.09261","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:a15881b2f333b3caa9abe7d82c330ba7b14b1f3ffe681b45be07d847e48dc646","observation_id":"825355f2-d67c-41ec-a68a-825ff5a528a9","resolution":{"observed_at":"2026-05-11T15:51:05.307004Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":"2403.08295","doi":"10.48550/arxiv.2403.08295","metadata_source":"pith","pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemma: Open Models Based on Gemini Research and Technology","venue":"cs.CL","work_id":"a9ea2870-df28-40b8-a9e0-a7e9a116f793","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:e710255ddea7fe0069c39ad4dbd5b849ca2ef929a74fdc94549035c87f6a3137","observation_id":"42704a8b-0d90-46d6-ac73-82d3f261e5e2","resolution":{"observed_at":"2026-05-11T15:51:05.511901Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":"2307.09288","doi":"10.24963/ijcai.2025/706","metadata_source":"pith","pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","venue":"cs.CL","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:3e07ed8390f8757c051a79e7e0eacd60ad0fccb6c4d93d423f2f976ba0455cdf","observation_id":"0e9b7867-776e-40cf-9c5d-817aec9f4813","resolution":{"observed_at":"2026-05-11T15:51:05.654440Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16944","last_updated":"2023-10-25T19:25:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-25T19:25:16Z","title":"Zephyr: Direct Distillation of LM Alignment","version":1},"cited_work":{"arxiv_id":"2310.16944","doi":"10.48550/arxiv.2310.16944","metadata_source":"pith","pith_arxiv_id":"2310.16944","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zephyr: Direct Distillation of LM Alignment","venue":"cs.LG","work_id":"ac74d5bf-9895-4253-9324-f50a8a6d9f20","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2310.16944","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:4908ed6baa2ecc81bc28bd6a5c8e1483e2817cbc5916fb57246c52b6cb5909e3","observation_id":"a54f8831-b313-4f41-8859-7733f6944d57","resolution":{"observed_at":"2026-05-16T10:13:57.632488Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1804.07461","last_updated":"2019-02-22T23:53:34Z","snapshot_observed_at":"2026-07-06T06:34:26.609892Z","submitted_at":"2018-04-20T06:35:04Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","version":3},"cited_work":{"arxiv_id":"1804.07461","doi":null,"metadata_source":"pith","pith_arxiv_id":"1804.07461","snapshot_observed_at":"2026-07-04T09:19:44.134394Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","venue":"cs.CL","work_id":"1bb6fb0c-482d-43cf-94a8-ed18f72a5563","year":2018},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/1804.07461","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:39914791113fe15d67094ffb863f9a51603a5ca1e7f9953a3e165167e5335dc7","observation_id":"9e9eebd7-178e-4594-85da-abfeecc9fe4e","resolution":{"observed_at":"2026-05-12T21:24:15.760169Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Superglue: A stickier benchmark for general-purpose language understanding systems","venue":null,"work_id":"3c76f631-bcf1-4abf-a501-962f28a91f17","year":2019},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:29574f9ea04ce89bd9105df6a20432b193e91ee46839819067256a42cd36d45f","observation_id":"d3cd9924-9ccf-47ac-b75b-2bdc04dfca44","resolution":{"observed_at":"2026-05-11T15:51:13.147828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11235","last_updated":"2024-03-16T04:32:25Z","snapshot_observed_at":"2026-07-06T16:21:13.175521Z","submitted_at":"2023-09-20T11:54:40Z","title":"OpenChat: Advancing Open-source Language Models with Mixed-Quality Data","version":2},"cited_work":{"arxiv_id":"2309.11235","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.11235","snapshot_observed_at":"2026-07-04T09:09:43.676147Z","title":"Openchat: Advancing open-source language models with mixed-quality data","venue":null,"work_id":"601dc112-6b35-4112-8a59-cc8115983a40","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2309.11235","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:d3f1a3a10457a94c23a03051d8ca5b051352ec17f891aae5918fce2e195a9d53","observation_id":"b9e16c97-0023-415e-b086-5c13d04767b6","resolution":{"observed_at":"2026-05-11T15:51:06.187905Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.10635","last_updated":"2024-06-28T08:24:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-20T07:01:57Z","title":"SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models","version":3},"cited_work":{"arxiv_id":"2307.10635","doi":"10.48550/arxiv.2307.10635","metadata_source":"pith","pith_arxiv_id":"2307.10635","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models","venue":"cs.CL","work_id":"88ae39fd-4d53-4184-9d82-03e8ac44d797","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2307.10635","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:0957820d1e03fd4ad3aea2e352c80fc34209e6baf768ff7ce9d0bb17154b567d","observation_id":"84c050a8-5ce5-40d8-ba90-e005cc210ad3","resolution":{"observed_at":"2026-05-16T12:59:45.390593Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-20T23:23:35.354385+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T23:23:35.354385+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":"d9139f92-75c3-4472-bd87-eb6cc680926b","year":2022},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:fdefb27215512e819bbc9cf511c13abf6b58104cb88e9d773a9c87c42329f15a","observation_id":"709bf6dd-02a4-460e-ab59-0ad70045c6eb","resolution":{"observed_at":"2026-05-11T15:51:13.407362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internlm-math: Open math large language models toward verifiable reasoning","venue":null,"work_id":"64bdd75e-54da-42fa-8e75-4d93eb07e77c","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:4a49eb8579e0f19f2b026f9f94837501c838277de7caf8ed66a458427faa667b","observation_id":"5d25fc4c-3820-4026-bafc-a91cb5f3456a","resolution":{"observed_at":"2026-05-11T15:51:13.546626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"cited_work":{"arxiv_id":"2403.04652","doi":"10.48550/arxiv.2403.04652","metadata_source":"pith","pith_arxiv_id":"2403.04652","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Yi: Open Foundation Models by 01.AI","venue":"cs.CL","work_id":"8efee8a1-5e3c-4851-9c65-18e3d1d9e769","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2403.04652","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:7d9c94b7bd195373c814034b142852595097b9252ecc8f5bd988763cfb054d3c","observation_id":"9247c9ee-d996-4f1e-b553-774a98c2fdfa","resolution":{"observed_at":"2026-05-13T05:47:28.044136Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03548","last_updated":"2024-05-23T16:34:35Z","snapshot_observed_at":"2026-08-06T17:52:58.973060Z","submitted_at":"2024-05-06T15:11:38Z","title":"MAmmoTH2: Scaling Instructions from the Web","version":4},"cited_work":{"arxiv_id":"2405.03548","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.03548","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mammoth2: Scaling instructions from the web","venue":null,"work_id":"94ee31c1-aec5-428f-8809-8cb5d9c313c9","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2405.03548","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:0e344042a735b2397149b25b72ceb373c1dd5ded748ce43287631ded5505d4e1","observation_id":"2b5358e2-adf0-4273-b65d-bdc78f894573","resolution":{"observed_at":"2026-05-11T15:51:06.893449Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.07830","last_updated":"2019-05-19T23:57:23Z","snapshot_observed_at":"2026-07-31T00:09:56.948833Z","submitted_at":"2019-05-19T23:57:23Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","version":1},"cited_work":{"arxiv_id":"1905.07830","doi":"10.48550/arxiv.1905.07830","metadata_source":"pith","pith_arxiv_id":"1905.07830","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","venue":"cs.CL","work_id":"79f44c0c-96f4-4edb-bc50-a3c9d6b85936","year":2019},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/1905.07830","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:9fd3798b31ad57cfe0f5dec8568a6506be95b6ca30bf9d215616b30cb11791ba","observation_id":"b8c5ff66-d693-44cc-b20b-0cb6569e2320","resolution":{"observed_at":"2026-05-11T15:51:07.104450Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Map-neo: Highly capable and transparent bilingual large language model series","venue":null,"work_id":"feb146e2-e4ab-4d8d-8c9b-c33edfc3ace4","year":2024},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:c28a4265961ed8b92dac3a18b106fb2cadf138a4a52f6752c9eace6b73fb5fb2","observation_id":"f7e6bb69-0747-4fa8-858c-abb009fa0f6b","resolution":{"observed_at":"2026-05-11T15:51:13.645427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Large language models are not robust multiple choice selectors","venue":null,"work_id":"c85bce6f-74fd-4ef3-8ea9-9080b533f4fc","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:434998312e0d39147d6a7dff838f70225f3da511d52f5952c37e5619fcce3331","observation_id":"3e6af77f-8187-4e43-9346-dcd1be80e88d","resolution":{"observed_at":"2026-05-11T15:51:13.874479Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06364","last_updated":"2023-09-18T14:23:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-13T09:39:30Z","title":"AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models","version":2},"cited_work":{"arxiv_id":"2304.06364","doi":"10.48550/arxiv.2304.06364","metadata_source":"pith","pith_arxiv_id":"2304.06364","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AGIEval: A Human-Centric Benchmark for Evaluating Foundation Models","venue":"cs.CL","work_id":"d42c58d5-eeb0-462a-92ab-3081ee269e59","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"cited_paper":"/paper/2304.06364","citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:1f6b6db1ef74f7c26bd8853a1402c7e48a494650d463a6730bed36758b681c73","observation_id":"9b4d48db-aefb-4c5d-8d47-bff912511c6e","resolution":{"observed_at":"2026-05-16T10:03:59.570579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"image_question","venue":null,"work_id":"f9128a55-7fca-4d77-aa03-b11ce5b9de01","year":2023},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:482940a14b074eb7facbecb6af0433285b87ac858b556adda02f7b5d0d010dea","observation_id":"8fc1faa6-3370-47a2-a4fd-c42d34a39ac0","resolution":{"observed_at":"2026-05-11T15:51:14.015975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Strain II has an average weight of 15 grams","venue":null,"work_id":"d9a1c27d-63c6-4320-978e-6c1a1e64f729","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:b6c2538088807010cedfa048824c494587bc69ca69fb1b300018b5edd83e7fac","observation_id":"85e82a9f-568c-42d5-8a63-0f8a970a3490","resolution":{"observed_at":"2026-05-11T15:51:14.224331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Each lowercase letter gene (a, b, d) contributes 2.5 grams","venue":null,"work_id":"2e4ad55d-4436-48e5-a336-c58c296279a7","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:9ed6c06028c46c29676726da7942592a90394c5e1a8dff2913cd53dccb758da1","observation_id":"c7efa464-4fb1-41be-bd4d-551bbc153f6a","resolution":{"observed_at":"2026-05-11T15:51:14.336253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- For Strain II (15 grams): - 15 = ( a + a + b + b + d + d) - Each lowercase letter contributes 2.5 grams, so: - 15 = 6 × 2.5 - Therefore, Strain II must have the genotype aabbdd","venue":null,"work_id":"83c0ea14-329c-4ead-9764-6e55abf99f58","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:058ebad2ef0427ff819ee5321e23ac6b7d6ece337a3146a0abed2c9b771a0eef","observation_id":"d53d232d-80ba-4113-8ee5-618e0d3f717e","resolution":{"observed_at":"2026-05-11T15:51:14.343457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e599684b-d935-4525-8ab4-0b13a58dd249","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:752425742a48d93592af01f1ab10b48530a873bd2ac940f551370f1ab9625041","observation_id":"96060590-f1cb-401b-bfe7-45e8157c048f","resolution":{"observed_at":"2026-05-11T15:51:14.454366Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"- Since there are three pairs, the total weight is 11.25 × 3 = 33 .75 grams","venue":null,"work_id":"2571f6eb-ca59-4170-9186-2496c56b3b8d","year":null},"citing_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-11T15:51:04.674346Z"},"links":{"citing_paper":"/paper/2406.01574"},"observation_digest":"sha256:5a2de2ba9c4189ed9950a0af9e3f8fad9f96f75934902efe3cc8feb9f2373ede","observation_id":"9b440d32-79a4-4cc9-9cf6-414074ad9ce3","resolution":{"observed_at":"2026-05-11T15:51:10.096972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","latest_version":6,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":1,"verified_exact":30,"verified_fuzzy":22},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 100 inbound Pith citation observations for arXiv:2406.01574."}