{"as_of":"2026-08-21T14:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6af50292208d9893ac64cb9f85f83a2ef7bd3a31d1a1bc2009f0556b9e22b9c5","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":49,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":49,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":49,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:12:43.047153Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":3,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-12T17:40:11.799681Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.12364","last_updated":"2025-02-06T09:36:58Z","snapshot_observed_at":"2026-08-20T12:51:23.149888Z","submitted_at":"2024-11-19T09:24:34Z","title":"Ultra-Sparse Memory Network","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-12T17:40:11.799681Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2411.12364"},"observation_digest":"sha256:849fe5bec993e4efbec3e95c83ce25abd793f947408b84570855b1bcf4f3b86a","observation_id":"55220a9d-b5c6-428f-b64d-973a86cb10ba","resolution":{"observed_at":"2026-08-12T17:40:11.799681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-11T12:03:42.589351Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14711","last_updated":"2025-02-27T16:33:09Z","snapshot_observed_at":"2026-08-14T22:43:45.227916Z","submitted_at":"2024-12-19T10:21:20Z","title":"ReMoE: Fully Differentiable Mixture-of-Experts with ReLU Routing","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T12:03:42.589351Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2412.14711"},"observation_digest":"sha256:5ac9434db561c96017ce16255a3115e7f99e1eeff9b8e2575267a6cb0cbae534","observation_id":"31f528f0-65bf-423d-9614-b46db656d994","resolution":{"observed_at":"2026-08-11T12:03:42.589351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-10T00:43:28.952649Z","title":"Scaling laws for fine-grained mixture of experts.arXiv preprint arXiv:2402.07871,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.18107","last_updated":"2025-06-07T00:03:08Z","snapshot_observed_at":"2026-08-13T19:29:56.924741Z","submitted_at":"2025-01-30T03:16:44Z","title":"Scaling Inference-Efficient Language Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T00:43:28.952649Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2501.18107"},"observation_digest":"sha256:60567d99c8f3fe58ae6c48167d03b151faa0918e929a615d78dea2e905cdbb1b","observation_id":"743d0cea-2e11-4eb2-99ef-552c3de26c33","resolution":{"observed_at":"2026-08-10T00:43:28.952649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-09T14:27:57.493957Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01804","last_updated":"2025-02-03T20:33:20Z","snapshot_observed_at":"2026-08-16T05:04:07.281702Z","submitted_at":"2025-02-03T20:33:20Z","title":"Soup-of-Experts: Pretraining Specialist Models via Parameters Averaging","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-09T14:27:57.493957Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2502.01804"},"observation_digest":"sha256:73829107e3a672a88bd23d6ff86210e78ffbe5576de8510a109da9dd0bf0fa4c","observation_id":"2745ea1f-7427-405e-851a-79a3be989fa8","resolution":{"observed_at":"2026-08-09T14:27:57.493957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-09T10:21:00.789646Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.03009","last_updated":"2025-06-16T05:27:08Z","snapshot_observed_at":"2026-08-16T01:06:03.699502Z","submitted_at":"2025-02-05T09:11:13Z","title":"Scaling Laws for Upcycling Mixture-of-Experts Language Models","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-09T10:21:00.789646Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2502.03009"},"observation_digest":"sha256:44389d4f6b5045e8520ba392f0acbccb5d295eda24d65036d3bb729368260c60","observation_id":"d2574811-c62c-4e89-af7d-ca2f6aad0326","resolution":{"observed_at":"2026-08-09T10:21:00.789646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-08T11:20:23.633073Z","title":"Kusupati, A., Bhatt, G., Rege, A., Wallingford, M., Sinha, A., Ramanujan, V ., Howard-Snyder, W., Chen, K., Kakade, S., Jain, P., and Farhadi, A","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.07972","last_updated":"2025-03-09T19:39:00Z","snapshot_observed_at":"2026-08-14T18:27:41.930983Z","submitted_at":"2025-02-11T21:36:31Z","title":"Training Sparse Mixture Of Experts Text Embedding Models","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T11:20:23.633073Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2502.07972"},"observation_digest":"sha256:ab6b1ef6c7b2606e5c2715dcd74ede0fab91bb63fb0510c3673fb25886f675a7","observation_id":"55cd9535-3e4b-4db5-aaa9-f9f4f0b56bfc","resolution":{"observed_at":"2026-08-08T11:20:23.633073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-16T10:12:43.047153Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18929","last_updated":"2025-04-26T14:02:07Z","snapshot_observed_at":"2026-08-18T12:12:00.004360Z","submitted_at":"2025-04-26T14:02:07Z","title":"Revisiting Transformers through the Lens of Low Entropy and Dynamic Sparsity","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-16T10:12:43.047153Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2504.18929"},"observation_digest":"sha256:28f9a5fa7c9919a24619f5d238b02e63820ee81cef093f3913e9124a8e5a5b68","observation_id":"dfb99475-540a-47e6-87b4-efed5b4f22c7","resolution":{"observed_at":"2026-08-16T10:12:43.047153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-16T04:39:14.062849Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00985","last_updated":"2025-05-25T14:53:34Z","snapshot_observed_at":"2026-08-17T19:26:39.886279Z","submitted_at":"2025-05-02T04:13:27Z","title":"Position: Enough of Scaling LLMs! Lets Focus on Downscaling","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T04:39:14.062849Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2505.00985"},"observation_digest":"sha256:955af78f54ceab549863bb82c7d0cc385baa19f22a1ae58f24cac0fdbf098718","observation_id":"9fbe071d-ce53-49c7-a60f-8c837cfbaf15","resolution":{"observed_at":"2026-08-16T04:39:14.062849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-15T22:46:44.587867Z","title":"Scaling laws for fine-grained mixture of experts.arXiv preprint arXiv:2402.07871,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06839","last_updated":"2025-05-11T04:35:40Z","snapshot_observed_at":"2026-08-18T12:12:02.040339Z","submitted_at":"2025-05-11T04:35:40Z","title":"The power of fine-grained experts: Granularity boosts expressivity in Mixture of Experts","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-15T22:46:44.587867Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2505.06839"},"observation_digest":"sha256:34710e34c01c3c108cacc612c7c348b062bcfc728ba0b53ea76047291804e320","observation_id":"cb3e4a9c-6470-4949-a038-3f529b28ba10","resolution":{"observed_at":"2026-08-15T22:46:44.587867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-07T14:34:13.881134Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18451","last_updated":"2025-05-24T01:23:02Z","snapshot_observed_at":"2026-08-19T07:03:03.481638Z","submitted_at":"2025-05-24T01:23:02Z","title":"$\\mu$-MoE: Test-Time Pruning as Micro-Grained Mixture-of-Experts","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T14:34:13.881134Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2505.18451"},"observation_digest":"sha256:0a951ba3e70b60e2ddf5f95101d6ee05755ae813792859061e5c2e9bc22bb699","observation_id":"6c3b664d-2096-4764-b95b-9e2381eac861","resolution":{"observed_at":"2026-08-07T14:34:13.881134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-07T11:18:59.617737Z","title":"arXiv: 2402.07871 [cs.LG]","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02890","last_updated":"2025-06-03T13:55:48Z","snapshot_observed_at":"2026-08-20T21:19:00.758198Z","submitted_at":"2025-06-03T13:55:48Z","title":"Scaling Fine-Grained MoE Beyond 50B Parameters: Empirical Evaluation and Practical Insights","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:18:59.617737Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2506.02890"},"observation_digest":"sha256:189c6d243496ec7c6f3590a88c54f1a1678fb83c92bf1b9896d9f1250367e496","observation_id":"49c2378d-5ed7-4d2f-a1f9-9e08474a9558","resolution":{"observed_at":"2026-08-07T11:18:59.617737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-07T05:07:39.916211Z","title":"Scaling laws for fine-grained mixture of experts,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00004","last_updated":"2025-07-10T17:08:40Z","snapshot_observed_at":"2026-08-17T13:53:41.674492Z","submitted_at":"2025-06-10T14:47:48Z","title":"A Theory of Inference Compute Scaling: Reasoning through Directed Stochastic Skill Search","version":2},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-08-07T05:07:39.916211Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2507.00004"},"observation_digest":"sha256:48d76772cf3ac526aa3cbddc1f35db612cc67254d19503aa7c1d0a559d8d8fe3","observation_id":"98c88a95-962e-476d-851c-d2b5d70c0932","resolution":{"observed_at":"2026-08-07T05:07:39.916211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2507.00029","last_updated":"2026-05-13T05:28:39Z","snapshot_observed_at":"2026-08-11T06:15:13.326412Z","submitted_at":"2025-06-17T14:58:54Z","title":"LoRA-Mixer: Coordinate Modular LoRA Experts Through Serial Attention Routing","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-19T09:05:25.236355Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2507.00029"},"observation_digest":"sha256:366018f7b1aee847bf9a9808aec575f71b633c1ee6aca3467bc41586c27ea8be","observation_id":"400a9e8f-9b73-45ca-8fbd-4df819028559","resolution":{"observed_at":"2026-05-19T09:07:14.578976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-06T18:20:00.225860Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08771","last_updated":"2025-07-30T04:14:15Z","snapshot_observed_at":"2026-08-15T02:00:29.298537Z","submitted_at":"2025-07-11T17:28:56Z","title":"BlockFFN: Towards End-Side Acceleration-Friendly Mixture-of-Experts with Chunk-Level Activation Sparsity","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T18:20:00.225860Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2507.08771"},"observation_digest":"sha256:218cd88207addcdcc0675acffde18a85c43c0c1f77acf4b6d51426349d3a8476","observation_id":"7060e794-53d6-4614-9403-522dba46b907","resolution":{"observed_at":"2026-08-06T18:20:00.225860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T19:23:15.826161Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.12801","last_updated":"2025-08-18T10:25:42Z","snapshot_observed_at":"2026-08-06T02:58:00.222739Z","submitted_at":"2025-08-18T10:25:42Z","title":"Maximum Score Routing For Mixture-of-Experts","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T19:23:15.826161Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2508.12801"},"observation_digest":"sha256:795c58356107d62fa19ad8baa975ee79762ac9876d123cf5c5fc78d1aa0651fa","observation_id":"7cb64e95-77f8-483e-8777-90264f295ad8","resolution":{"observed_at":"2026-08-05T19:23:15.826161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T16:17:42.037489Z","title":"Scaling laws for fine-grained mixture of experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.18756","last_updated":"2025-08-26T07:33:11Z","snapshot_observed_at":"2026-08-20T23:44:48.149491Z","submitted_at":"2025-08-26T07:33:11Z","title":"UltraMemV2: Memory Networks Scaling to 120B Parameters with Superior Long-Context Learning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T16:17:42.037489Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2508.18756"},"observation_digest":"sha256:cb50ec3e738f4d4f010e00a37d3e7f02200beaa5681cc19bc00a1a44a48a6eb0","observation_id":"e766e8a8-41d6-4b09-a7bd-4f0b18b05e96","resolution":{"observed_at":"2026-08-05T16:17:42.037489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2509.19349","last_updated":"2025-09-17T17:49:02Z","snapshot_observed_at":"2026-08-15T22:10:58.636387Z","submitted_at":"2025-09-17T17:49:02Z","title":"ShinkaEvolve: Towards Open-Ended And Sample-Efficient Program Evolution","version":1},"reference_index":163,"source":"arxiv_source","source_observed_at":"2026-05-16T13:58:58.627748Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2509.19349"},"observation_digest":"sha256:d6545600f96229f12934f23a754f1adc084608fb0aa5ac3e1d4ba999b6ac4e48","observation_id":"9b19e677-ad68-4a5e-a034-715ac56ca56c","resolution":{"observed_at":"2026-05-16T13:58:58.957195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2510.18245","last_updated":"2026-05-13T04:16:31Z","snapshot_observed_at":"2026-08-14T06:24:23.080268Z","submitted_at":"2025-10-21T03:08:48Z","title":"Scaling Laws Meet Model Architecture: Toward Inference-Efficient LLMs","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T05:30:11.389756Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2510.18245"},"observation_digest":"sha256:19479ac888ea9d4fa9bfab7983e588cd9c78625fca20b91b28259f25b0418005","observation_id":"8e35aa42-8db1-4f09-a7d3-54e90d3459ab","resolution":{"observed_at":"2026-05-18T05:30:55.127020Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-02T21:50:58.392342Z","title":"Scaling laws for fine-grained mixture of experts.arXiv preprint arXiv:2402.07871,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.06626","last_updated":"2026-05-24T09:37:35Z","snapshot_observed_at":"2026-08-17T12:24:56.896194Z","submitted_at":"2026-02-22T06:09:57Z","title":"Grouter: Decoupling Routing from Representation for Accelerated MoE Training","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T21:50:58.392342Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2603.06626"},"observation_digest":"sha256:e06855925639e81abce5af572f2e11a6927f3c14274a2d533e947e9e51548899","observation_id":"575042b2-3f11-4599-9691-8941c8874a45","resolution":{"observed_at":"2026-08-02T21:50:58.392342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2604.09175","last_updated":"2026-04-10T09:59:48Z","snapshot_observed_at":"2026-08-13T19:58:30.790831Z","submitted_at":"2026-04-10T09:59:48Z","title":"Generalization and Scaling Laws for Mixture-of-Experts Transformers","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T18:20:45.939113Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2604.09175"},"observation_digest":"sha256:24e8dfb350d2f66292bf43ed38bdb68c64f0948f3ed9592e755e3e383d931a19","observation_id":"e754b830-dc22-4c6f-97a1-942d36202bfa","resolution":{"observed_at":"2026-05-10T20:25:46.779709Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2604.18473","last_updated":"2026-04-20T16:24:41Z","snapshot_observed_at":"2026-08-14T20:51:20.111244Z","submitted_at":"2026-04-20T16:24:41Z","title":"Train Separately, Merge Together: Modular Post-Training with Mixture-of-Experts","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-10T05:52:28.822723Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2604.18473"},"observation_digest":"sha256:e28ea6c8b84508f5e015ff3bdd9871d9aaf47f0c98464ab61aafc4176545a336","observation_id":"ceef38a6-2501-4310-8074-12b52e4e37c8","resolution":{"observed_at":"2026-05-10T05:56:11.386751Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2604.19835","last_updated":"2026-05-10T18:33:52Z","snapshot_observed_at":"2026-08-15T19:42:17.129212Z","submitted_at":"2026-04-21T05:53:33Z","title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T03:29:16.555166Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2604.19835"},"observation_digest":"sha256:0e12eb2f4cde2762736fc37c8f4a169cbde902b78a45c8153d185e3fadc3f642","observation_id":"d8efa6c9-f131-4c80-aa40-b8c5b9c1904b","resolution":{"observed_at":"2026-05-10T03:29:21.467504Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2604.19835","last_updated":"2026-05-10T18:33:52Z","snapshot_observed_at":"2026-08-15T19:42:17.129212Z","submitted_at":"2026-04-21T05:53:33Z","title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T02:03:02.654035Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2604.19835"},"observation_digest":"sha256:b05ca359dfdbca13c1b4d08dbb801b7377bac2c1b9148a6f3b323c172ef2bf28","observation_id":"7b0baec4-ce32-48b3-bd14-064caa41a3b2","resolution":{"observed_at":"2026-05-12T02:06:15.321303Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2604.21691","last_updated":"2026-04-23T13:58:12Z","snapshot_observed_at":"2026-08-02T12:48:50.592713Z","submitted_at":"2026-04-23T13:58:12Z","title":"There Will Be a Scientific Theory of Deep Learning","version":1},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-05-09T20:11:17.616190Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2604.21691"},"observation_digest":"sha256:de822f37012b995226d23ecb8fbd22906e5f6a65fa9a39b57986064aebf3b9f9","observation_id":"45123722-e5eb-4cf8-af05-7f7c9e722391","resolution":{"observed_at":"2026-05-11T15:21:09.187987Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2604.24037","last_updated":"2026-06-22T01:13:01Z","snapshot_observed_at":"2026-07-06T23:10:12.016186Z","submitted_at":"2026-04-27T04:43:42Z","title":"A Limit Theory of Foundation Models: A Mathematical Approach to Understanding Emergent Intelligence and Scaling Laws","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-13T07:27:21.118156Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2604.24037"},"observation_digest":"sha256:9bf14efc3765719fdff0b92d52388d3d811864abbadea395652a964b583fe5fa","observation_id":"1ba0a64a-8f23-47be-8c72-6226476b4d19","resolution":{"observed_at":"2026-05-13T07:27:28.919417Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2604.24037","last_updated":"2026-06-22T01:13:01Z","snapshot_observed_at":"2026-07-06T23:10:12.016186Z","submitted_at":"2026-04-27T04:43:42Z","title":"A Limit Theory of Foundation Models: A Mathematical Approach to Understanding Emergent Intelligence and Scaling Laws","version":4},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-07-01T09:03:59.522516Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2604.24037"},"observation_digest":"sha256:8a1eb80872af639587a2dc337272455efa2cf4f11412b6bfeff4e864dbb50492","observation_id":"6f547a1d-5a89-4010-b906-12d4e5ea99e7","resolution":{"observed_at":"2026-07-01T09:05:36.307668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.06665","last_updated":"2026-05-07T17:59:44Z","snapshot_observed_at":"2026-08-11T16:17:43.420898Z","submitted_at":"2026-05-07T17:59:44Z","title":"UniPool: A Globally Shared Expert Pool for Mixture-of-Experts","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-08T11:56:21.623709Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.06665"},"observation_digest":"sha256:0d41eea46d019184947704288b510534e6faea4d0ea4629f075662ecf09759b2","observation_id":"1e64b776-ec3c-4bba-9cdc-53598699f42b","resolution":{"observed_at":"2026-05-11T19:26:09.346763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.10933","last_updated":"2026-05-20T17:26:14Z","snapshot_observed_at":"2026-08-16T09:01:03.343803Z","submitted_at":"2026-05-11T17:58:28Z","title":"DECO: Sparse Mixture-of-Experts with Dense-Comparable Performance on End-Side Devices","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-05-12T03:36:12.915133Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.10933"},"observation_digest":"sha256:f7edadf936f7efb3c9ac5031e3707341093114d02d01c6a8ddf9515f1585dd9e","observation_id":"fa55a180-7d94-41e7-868d-5485dca7be36","resolution":{"observed_at":"2026-05-12T03:36:20.061044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.10933","last_updated":"2026-05-20T17:26:14Z","snapshot_observed_at":"2026-08-16T09:01:03.343803Z","submitted_at":"2026-05-11T17:58:28Z","title":"DECO: Sparse Mixture-of-Experts with Dense-Comparable Performance on End-Side Devices","version":2},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-05-13T07:29:14.545746Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.10933"},"observation_digest":"sha256:4e9b172fde74080b7082f183d566c47260dd21b9bf07196e6d003273ae8136c9","observation_id":"820d46e4-9493-4e7b-9dee-e5422aa56357","resolution":{"observed_at":"2026-05-13T07:32:30.284812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.10933","last_updated":"2026-05-20T17:26:14Z","snapshot_observed_at":"2026-08-16T09:01:03.343803Z","submitted_at":"2026-05-11T17:58:28Z","title":"DECO: Sparse Mixture-of-Experts with Dense-Comparable Performance on End-Side Devices","version":3},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-05-21T07:57:49.746594Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.10933"},"observation_digest":"sha256:83c5cc96c7dd82e7549786951b0354e3fa13886bb17042faf923a2fadd2acfa4","observation_id":"d8f2b55c-a351-4ca9-8d83-0129865420f0","resolution":{"observed_at":"2026-05-21T07:59:50.244456Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.12715","last_updated":"2026-05-15T17:01:07Z","snapshot_observed_at":"2026-07-06T23:24:23.402304Z","submitted_at":"2026-05-12T20:22:45Z","title":"Scaling Laws for Mixture Pretraining Under Data Constraints","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-14T21:44:31.429223Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.12715"},"observation_digest":"sha256:e6686aa4b40745b31219506361591e269ccf8ca3a0071b5f37730246d4ff2006","observation_id":"fc4069e4-7fca-491a-9c8c-2be0f7349e8b","resolution":{"observed_at":"2026-05-14T21:48:00.974426Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.12715","last_updated":"2026-05-15T17:01:07Z","snapshot_observed_at":"2026-07-06T23:24:23.402304Z","submitted_at":"2026-05-12T20:22:45Z","title":"Scaling Laws for Mixture Pretraining Under Data Constraints","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-19T16:36:30.007014Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.12715"},"observation_digest":"sha256:b3160e5acd7ad2967d0b397d7eeb43cc40c494437bb8e15f81a8b75e8d1f0846","observation_id":"5159d113-4454-4e88-8f60-7fdc40fdf00a","resolution":{"observed_at":"2026-05-19T16:37:39.587997Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.13769","last_updated":"2026-05-13T16:48:24Z","snapshot_observed_at":"2026-08-17T02:54:40.227749Z","submitted_at":"2026-05-13T16:48:24Z","title":"Dense vs Sparse Pretraining at Tiny Scale: Active-Parameter vs Total-Parameter Matching","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-14T19:18:56.831844Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.13769"},"observation_digest":"sha256:2f4985bb6c2f909cd4551f06230e3fb569a2b9a119bd22011f23a9990066323c","observation_id":"6bf453b0-12b7-4e14-b8f2-f9317e9ebfba","resolution":{"observed_at":"2026-05-14T19:19:23.630198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.14200","last_updated":"2026-05-13T23:32:00Z","snapshot_observed_at":"2026-07-06T23:25:39.415637Z","submitted_at":"2026-05-13T23:32:00Z","title":"How to Scale Mixture-of-Experts: From muP to the Maximally Scale-Stable Parameterization","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-15T04:45:20.091598Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.14200"},"observation_digest":"sha256:2fa740283c8ae7b0c06d3823494a8cb1e23d9aad7e41611244d703dffcced27b","observation_id":"213053b1-55a7-4c84-8c28-dd2d5debf6e1","resolution":{"observed_at":"2026-05-15T04:49:44.995043Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.16349","last_updated":"2026-05-08T04:17:10Z","snapshot_observed_at":"2026-08-15T07:08:51.815605Z","submitted_at":"2026-05-08T04:17:10Z","title":"Geometric Asymmetry in MoE Specialization: Functional Decorrelation and Representational Overlap","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-20T23:45:19.279268Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.16349"},"observation_digest":"sha256:fdfac62adfc39096ec598b4234cd7b86318b3aa125345d12a49530fb265dc177","observation_id":"d2d51c58-8a96-4053-880a-5cc1292b81ac","resolution":{"observed_at":"2026-05-20T23:49:15.472406Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.27358","last_updated":"2026-05-26T17:58:24Z","snapshot_observed_at":"2026-08-12T17:35:26.802008Z","submitted_at":"2026-05-26T17:58:24Z","title":"MobileMoE: Scaling On-Device Mixture of Experts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T18:48:50.656971Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.27358"},"observation_digest":"sha256:156a6b6c683bc18acc75395b803b2222ff78b6efa249f56b91ae9b80edff2956","observation_id":"24c71126-e941-4afb-b4ce-9a5237762316","resolution":{"observed_at":"2026-06-29T18:53:51.428104Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.27722","last_updated":"2026-05-26T21:49:16Z","snapshot_observed_at":"2026-08-10T10:26:14.129729Z","submitted_at":"2026-05-26T21:49:16Z","title":"NUCLEUS-MoE: Unified Model of Pool Boiling for Liquid Cooling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-29T18:30:09.813448Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.27722"},"observation_digest":"sha256:a871bde8b12d6d3bd87931f104aa645aab6bf13503bc77266b105bb5af246e00","observation_id":"d0415849-9705-47e4-b36e-c54fdcd1b953","resolution":{"observed_at":"2026-06-29T18:33:50.529632Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2605.31268","last_updated":"2026-05-29T13:01:11Z","snapshot_observed_at":"2026-08-12T23:02:09.228331Z","submitted_at":"2026-05-29T13:01:11Z","title":"Mellum2 Technical Report","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-28T22:58:35.397914Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2605.31268"},"observation_digest":"sha256:895269572ebf38c0747671a7d6d835b99cf49b02abbd554ed0b7bd028253ce7d","observation_id":"a8bf0ac9-82c1-4dc2-b71c-6541060d964b","resolution":{"observed_at":"2026-06-28T23:02:46.509443Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2606.07414","last_updated":"2026-06-05T16:06:47Z","snapshot_observed_at":"2026-08-17T17:53:01.281961Z","submitted_at":"2026-06-05T16:06:47Z","title":"Sparsely gated tiny linear experts","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T22:49:49.299925Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2606.07414"},"observation_digest":"sha256:4f5d253923e62e0695677ff1bd0f248df9c220dc9856b7ca5cb5e404c5decc46","observation_id":"ceebb710-eee8-4251-b3d4-61676104f119","resolution":{"observed_at":"2026-07-02T16:17:09.328956Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2606.21228","last_updated":"2026-06-23T04:40:39Z","snapshot_observed_at":"2026-08-11T18:07:40.444367Z","submitted_at":"2026-06-19T08:47:40Z","title":"Sakana Fugu Technical Report","version":2},"reference_index":197,"source":"arxiv_source","source_observed_at":"2026-06-26T14:22:37.596720Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2606.21228"},"observation_digest":"sha256:539ca1848a53f1af7459aabecf6c97a6227d382ab66f90dc649b35903096ba26","observation_id":"bcf148dd-48f2-4345-a0c2-b446fb0ff84c","resolution":{"observed_at":"2026-07-04T06:29:38.264146Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2402.07871","doi":"10.48550/arxiv.2402.07871","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":"arXiv (Cornell University)","work_id":"e67733fa-7550-4e7a-b1e0-d65341a18264","year":2024},"citing_paper":{"arxiv_id":"2606.31397","last_updated":"2026-06-30T09:25:37Z","snapshot_observed_at":"2026-08-03T17:48:59.429755Z","submitted_at":"2026-06-30T09:25:37Z","title":"Mixture-of-Control: State-Aware Fine-Tuning for Transformer-based Models","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-07-01T06:55:06.270685Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2606.31397"},"observation_digest":"sha256:e0c98ac89706ddf563e5f0872ba78c89898afa51c6e11c1d4f19a4057ab62fba","observation_id":"7468852c-78b1-402e-be66-805467163583","resolution":{"observed_at":"2026-07-01T06:55:28.676268Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-07-13T03:06:27.991558Z","title":"Krajewski, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09424","last_updated":"2026-07-22T18:23:55Z","snapshot_observed_at":"2026-08-13T16:27:07.974362Z","submitted_at":"2026-07-10T13:51:41Z","title":"A Sovereign, Open-Source Foundation Model for German and English","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-13T03:06:27.991558Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2607.09424"},"observation_digest":"sha256:510bb4a7e69cb9c2300a53855c1a3088ec25d55839af18e4fd5c4c62f8235f2d","observation_id":"144fa094-ef37-4303-9c48-495b7e44c7f4","resolution":{"observed_at":"2026-07-13T03:06:27.991558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-07-14T15:13:39.458378Z","title":"Krajewski, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09424","last_updated":"2026-07-22T18:23:55Z","snapshot_observed_at":"2026-08-13T16:27:07.974362Z","submitted_at":"2026-07-10T13:51:41Z","title":"A Sovereign, Open-Source Foundation Model for German and English","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-14T15:13:39.458378Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2607.09424"},"observation_digest":"sha256:7f000f1b00736ddc36088aef00aed605eb515d0e695a5f0c8208ac345d7bd6f0","observation_id":"0f8e0912-303f-47ed-955e-6cf128de49da","resolution":{"observed_at":"2026-07-14T15:13:39.458378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-02T07:45:33.911404Z","title":"Krajewski, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09424","last_updated":"2026-07-22T18:23:55Z","snapshot_observed_at":"2026-08-13T16:27:07.974362Z","submitted_at":"2026-07-10T13:51:41Z","title":"A Sovereign, Open-Source Foundation Model for German and English","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T07:45:33.911404Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2607.09424"},"observation_digest":"sha256:08df2ac85e1f17b63c78d7d2410ce9c10dc3ff0454ac11f46bd6c83dfe2b6f0e","observation_id":"6d1534c2-4c77-457b-a38c-41fdcbecf3a9","resolution":{"observed_at":"2026-08-02T07:45:33.911404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-01T10:42:50.252606Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.20145","last_updated":"2026-08-19T09:30:55Z","snapshot_observed_at":"2026-08-21T13:13:18.061147Z","submitted_at":"2026-07-22T13:49:17Z","title":"SLAI T-Rex: Full-Parameter Post-training of the DeepSeek-V4 Family on Ascend SuperPOD","version":2},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-08-01T10:42:50.252606Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2607.20145"},"observation_digest":"sha256:ca6c7acd11370bc6ef679babb3c7ef5257f01c2a39f19618cabda7d0a76297c4","observation_id":"a0086c4a-66db-400a-9d47-6c838b4a1820","resolution":{"observed_at":"2026-08-01T10:42:50.252606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-02T14:39:32.890858Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.20427","last_updated":"2026-05-08T16:55:49Z","snapshot_observed_at":"2026-08-20T19:10:05.195026Z","submitted_at":"2026-05-08T16:55:49Z","title":"Is MoE Routing a Huffman Code? Discovering the Frequency-Diversity Law in Chain-of-Thought","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-02T14:39:32.890858Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2607.20427"},"observation_digest":"sha256:585395a2f3140fd2699b29d39a277740c1eeb85a1684334333de14e57cbcec3c","observation_id":"1726809c-d790-4fd0-8282-ceed4e3e41ef","resolution":{"observed_at":"2026-08-02T14:39:32.890858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-07-30T12:53:41.147044Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23777","last_updated":"2026-07-26T17:55:40Z","snapshot_observed_at":"2026-07-30T23:56:43.497541Z","submitted_at":"2026-07-26T17:55:40Z","title":"Scale Weight Decay and Train Better","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-07-30T12:53:41.147044Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2607.23777"},"observation_digest":"sha256:0ebdbf9cf95b5428bf9a1a966f21de1f9257e2636d9effa7f888524b7a0b1179","observation_id":"80569cfd-2968-4e13-864e-cd06e2a97b27","resolution":{"observed_at":"2026-07-30T12:53:41.147044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-15T14:54:53.172947Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03457","last_updated":"2026-08-04T10:53:02Z","snapshot_observed_at":"2026-08-20T21:57:29.994762Z","submitted_at":"2026-08-04T10:53:02Z","title":"LLaDA MoE v2: Scaling Mixture-of-Experts Diffusion Language Models","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-15T14:54:53.172947Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2608.03457"},"observation_digest":"sha256:40e54e7876ed725af54850659d6c61d97f9a70d5fe7bca6c8e52e0928e2b6d40","observation_id":"1bc886c4-2d67-4239-9e0a-ad01dd08c488","resolution":{"observed_at":"2026-08-15T14:54:53.172947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07871","snapshot_observed_at":"2026-08-12T21:08:06.453774Z","title":"arXiv preprint arXiv:2402.07871 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10605","last_updated":"2026-08-11T07:49:00Z","snapshot_observed_at":"2026-08-16T11:47:48.638035Z","submitted_at":"2026-08-11T07:49:00Z","title":"Compute-Optimal Is Not Cluster-Optimal: Systems-Aware Scaling for Sparse Mixture-of-Experts","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-12T21:08:06.453774Z"},"links":{"cited_paper":"/paper/2402.07871","citing_paper":"/paper/2608.10605"},"observation_digest":"sha256:02461584e63fc83638b1a1e71d93f694b6cebc109e0a761691478d7aa5136634","observation_id":"f82952f8-c475-4040-87f5-21e42320667a","resolution":{"observed_at":"2026-08-12T21:08:06.453774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.07871/citation-record","integrity":"/paper/2402.07871/integrity","json":"/paper/2402.07871/citation-record.json","paper":"/paper/2402.07871"},"outbound":[],"paper":{"arxiv_id":"2402.07871","last_updated":"2024-02-12T18:33:47Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-18T12:11:18.333774Z","submitted_at":"2024-02-12T18:33:47Z","title":"Scaling Laws for Fine-Grained Mixture of Experts"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 49 inbound Pith citation observations for arXiv:2402.07871."}