{"as_of":"2026-08-11T01:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:75cc6a3600826a543a1f527f636f644e05f4ea672d67c1acdfbf995101c73abb","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T13:40:31.736571Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:27:49.926091Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T05:27:51.662445Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"cited_work":{"arxiv_id":"2501.16295","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.16295","snapshot_observed_at":"2026-08-07T05:27:51.662445Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","venue":"cs.LG","work_id":"08573191-4761-40c1-9837-50945b695104","year":2025},"citing_paper":{"arxiv_id":"2506.07976","last_updated":"2025-06-10T12:50:18Z","snapshot_observed_at":"2026-08-07T20:50:35.036523Z","submitted_at":"2025-06-09T17:50:02Z","title":"Thinking vs. Doing: Agents that Reason by Scaling Test-Time Interaction","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:27:49.926091Z"},"links":{"cited_paper":"/paper/2501.16295","citing_paper":"/paper/2506.07976"},"observation_digest":"sha256:589386e343046e93d3554347bb3d4e4a3595a5ad5feb507cbd4de5b287ce1cbc","observation_id":"f13cb217-ff8e-47c5-a4bf-d67e8501f29f","resolution":{"observed_at":"2026-08-07T05:27:51.667553Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.16295/citation-record","integrity":"/paper/2501.16295/integrity","json":"/paper/2501.16295/citation-record.json","paper":"/paper/2501.16295"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.01771","last_updated":"2024-02-01T07:15:58Z","snapshot_observed_at":"2026-08-10T21:25:11.091970Z","submitted_at":"2024-02-01T07:15:58Z","title":"BlackMamba: Mixture of Experts for State-Space Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01771","snapshot_observed_at":"2026-08-10T13:40:31.658057Z","title":"Blackmamba: Mixture of experts for state-space models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.658057Z"},"links":{"cited_paper":"/paper/2402.01771","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:09c5b1ae5a03e61babbed242f570ac0b4226354acef8a859e58d768cf2f78900","observation_id":"e1e54f7f-f4be-46ba-bd09-0f5e4050a66d","resolution":{"observed_at":"2026-08-10T13:40:31.658057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09818","last_updated":"2025-03-21T05:54:00Z","snapshot_observed_at":"2026-08-09T20:05:31.409634Z","submitted_at":"2024-05-16T05:23:41Z","title":"Chameleon: Mixed-Modal Early-Fusion Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.09818","snapshot_observed_at":"2026-08-10T13:40:31.665609Z","title":"org/abs/2405.09818","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.665609Z"},"links":{"cited_paper":"/paper/2405.09818","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:cc9c563b6e83e2acd307418d68ba175e2a9eb99fa86a5ca0fb3753399eb63431","observation_id":"966c4c3c-53b9-4ea2-8352-04f941ff620c","resolution":{"observed_at":"2026-08-10T13:40:31.665609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.13131","last_updated":"2022-03-24T15:44:50Z","snapshot_observed_at":"2026-08-06T19:08:21.390272Z","submitted_at":"2022-03-24T15:44:50Z","title":"Make-A-Scene: Scene-Based Text-to-Image Generation with Human Priors","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.13131","snapshot_observed_at":"2026-08-10T13:40:31.672333Z","title":"Make-a-scene: Scene-based text- to-image generation with human priors","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.672333Z"},"links":{"cited_paper":"/paper/2203.13131","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:805f3999f48862a25962270fc7a00a5e587cf83fa7fddd6a83b5a67d0765481d","observation_id":"fa1eca84-31aa-4b5a-aa4d-8addbb37b868","resolution":{"observed_at":"2026-08-10T13:40:31.672333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-10T13:40:31.675443Z","title":"and Dao, T","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.675443Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:798bd5c8fd8c9a3f800e4e5f45ab9b4ead9996877014097b2c8de1f69ed98fe2","observation_id":"8099324c-7760-4942-859e-eda2129029a9","resolution":{"observed_at":"2026-08-10T13:40:31.675443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-08T06:16:25.839566Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-10T13:40:31.684781Z","title":"org/abs/2401.04088","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.684781Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:e4850754dd2e3f3fd3d562437f2ddd8bd641207497cfd2914048199bccf3b0dc","observation_id":"e0113b73-1045-4717-84db-9e43b3b38861","resolution":{"observed_at":"2026-08-10T13:40:31.684781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04996","last_updated":"2025-05-08T01:53:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T18:59:06Z","title":"Mixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04996","snapshot_observed_at":"2026-08-10T13:40:31.691269Z","title":"Mixture-of-transformers: A sparse and scalable architec- ture for multi-modal foundation models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.691269Z"},"links":{"cited_paper":"/paper/2411.04996","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:32386830eab8d1d80dea77901b366646652b4cd1b288f2f4b44266222945b4cd","observation_id":"f2a9dc46-5651-4177-8449-d4332f734b13","resolution":{"observed_at":"2026-08-10T13:40:31.691269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21770","last_updated":"2024-08-12T16:20:37Z","snapshot_observed_at":"2026-08-09T08:07:42.316346Z","submitted_at":"2024-07-31T17:46:51Z","title":"MoMa: Efficient Early-Fusion Pre-training with Mixture of Modality-Aware Experts","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21770","snapshot_observed_at":"2026-08-10T13:40:31.694571Z","title":"V ., Shrivastava, A., Luo, L., Iyer, S., Lewis, M., Gosh, G., Zettlemoyer, L., and Aghajanyan, A","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.694571Z"},"links":{"cited_paper":"/paper/2407.21770","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:c76cd3157d9d6708d5bf37f5b2a429ce3bbd4a1e9e8da257f2096d76b093cbf6","observation_id":"a1c2de6c-835e-4a82-aeb2-60512abf1740","resolution":{"observed_at":"2026-08-10T13:40:31.694571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10005","last_updated":"2024-01-16T05:43:20Z","snapshot_observed_at":"2026-07-06T15:28:32.440071Z","submitted_at":"2023-05-17T07:23:46Z","title":"DinoSR: Self-Distillation and Online Clustering for Self-supervised Speech Representation Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10005","snapshot_observed_at":"2026-08-10T13:40:31.697617Z","title":"H., Chang, H.-J., Auli, M., Hsu, W.-N., and Glass, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.697617Z"},"links":{"cited_paper":"/paper/2305.10005","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:825545760fb763fb4883c294f8f0d7922dff79ab877157f81b8cada003ed6262","observation_id":"0a70059c-3f42-41f5-86b0-cbb1009252f1","resolution":{"observed_at":"2026-08-10T13:40:31.697617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.15881","last_updated":"2024-05-24T18:50:27Z","snapshot_observed_at":"2026-08-05T12:38:35.366695Z","submitted_at":"2024-05-24T18:50:27Z","title":"Scaling Diffusion Mamba with Bidirectional SSMs for Efficient Image and Video Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.15881","snapshot_observed_at":"2026-08-10T13:40:31.700705Z","title":"and Tian, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.700705Z"},"links":{"cited_paper":"/paper/2405.15881","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:24f6c482f22f3c75ed865bb498499471414785d15d21f119690f06bd72f23028","observation_id":"2482b423-b6b8-466f-82ed-b59f64d72a31","resolution":{"observed_at":"2026-08-10T13:40:31.700705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13600","last_updated":"2024-03-20T13:48:50Z","snapshot_observed_at":"2026-08-10T13:01:51.483664Z","submitted_at":"2024-03-20T13:48:50Z","title":"VL-Mamba: Exploring State Space Models for Multimodal Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13600","snapshot_observed_at":"2026-08-10T13:40:31.704039Z","title":"Vl-mamba: Exploring state space models for multimodal learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.704039Z"},"links":{"cited_paper":"/paper/2403.13600","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:fe4a77f23a13d41d5b3f3e3911a8bed182f918ce247b4477444ea54ff1a5a947","observation_id":"738b8640-4580-4cb1-b3ef-3a9f2a5a2ea9","resolution":{"observed_at":"2026-08-10T13:40:31.704039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-08-10T13:40:31.707420Z","title":"Glu variants improve transformer","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.707420Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:b31d77efa5522e0406b2654c29f2a89a3076f447b95aba693e408b339289e54a","observation_id":"a5400b72-263e-45e9-900a-ea901e4fecd1","resolution":{"observed_at":"2026-08-10T13:40:31.707420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15004","last_updated":"2024-12-05T02:00:07Z","snapshot_observed_at":"2026-08-05T16:22:08.749834Z","submitted_at":"2024-11-22T15:26:23Z","title":"ScribeAgent: Towards Specialized Web Agents Using Production-Scale Workflow Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15004","snapshot_observed_at":"2026-08-10T13:40:31.717857Z","title":"M., Staten, C., Khodak, M., Neubig, G., and Talwalkar, A","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.717857Z"},"links":{"cited_paper":"/paper/2411.15004","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:c98c6fab63456769607a4a14269815ed28acb0366e7caced283340e1af69d71d","observation_id":"90168bb9-d3d3-41a1-9f32-d9750a85bdeb","resolution":{"observed_at":"2026-08-10T13:40:31.717857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07816","last_updated":"2024-03-12T16:54:58Z","snapshot_observed_at":"2026-08-10T12:02:37.923207Z","submitted_at":"2024-03-12T16:54:58Z","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07816","snapshot_observed_at":"2026-08-10T13:40:31.724880Z","title":"org/abs/2403.07816","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.724880Z"},"links":{"cited_paper":"/paper/2403.07816","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:9aba124ca0b672110470d34eb68b6386e49339d8bcbdfc1abc413f4b564f6743","observation_id":"1e1cd6be-58ee-4e87-a7d3-7c128ff908d7","resolution":{"observed_at":"2026-08-10T13:40:31.724880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2208.10442","last_updated":"2022-08-31T02:26:45Z","snapshot_observed_at":"2026-08-11T00:20:29.155513Z","submitted_at":"2022-08-22T16:55:04Z","title":"Image as a Foreign Language: BEiT Pretraining for All Vision and Vision-Language Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.10442","snapshot_observed_at":"2026-08-10T13:40:31.727960Z","title":"Wang, W., Lv, Q., Yu, W., Hong, W., Qi, J., Wang, Y ., Ji, J., Yang, Z., Zhao, L., Song, X., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.727960Z"},"links":{"cited_paper":"/paper/2208.10442","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:778964a1b07a945d846e34d8261cf65088b70311541b3833617d63b027d0724d","observation_id":"398e87de-fad2-47ad-bc3b-faba6250862c","resolution":{"observed_at":"2026-08-10T13:40:31.727960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.02796","last_updated":"2025-03-21T03:59:29Z","snapshot_observed_at":"2026-07-06T19:45:17.343568Z","submitted_at":"2024-11-05T04:10:59Z","title":"Specialized Foundation Models Struggle to Beat Supervised Baselines","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.02796","snapshot_observed_at":"2026-08-10T13:40:31.730810Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.730810Z"},"links":{"cited_paper":"/paper/2411.02796","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:0c7212a61bb69fb58d3d361f2f720949015a8a84c2929511751ca78f8e624455","observation_id":"6d64e3e0-bdfa-4af8-b635-676b2c3e3308","resolution":{"observed_at":"2026-08-10T13:40:31.730810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.14520","last_updated":"2025-01-08T11:03:00Z","snapshot_observed_at":"2026-07-06T17:48:24.076444Z","submitted_at":"2024-03-21T16:17:57Z","title":"Cobra: Extending Mamba to Multi-Modal Large Language Model for Efficient Inference","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.14520","snapshot_observed_at":"2026-08-10T13:40:31.733836Z","title":"Cobra: Extending mamba to multi-modal large language model for efficient inference","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.733836Z"},"links":{"cited_paper":"/paper/2403.14520","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:6308172442e0b0c864a51a483b256baa32063746d4fe066a579cdeee09aabd8f","observation_id":"bf36deb8-d066-4e15-9593-2faf5df2f54f","resolution":{"observed_at":"2026-08-10T13:40:31.733836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11039","last_updated":"2024-08-20T17:48:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-20T17:48:20Z","title":"Transfusion: Predict the Next Token and Diffuse Images with One Multi-Modal Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.11039","snapshot_observed_at":"2026-08-10T13:40:31.736571Z","title":"Transfusion: Predict the next token and dif- fuse images with one multi-modal model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.736571Z"},"links":{"cited_paper":"/paper/2408.11039","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:8fb57cd015f8dfc31602d9f592807e415beed84423f8da962cfd390c8645b111","observation_id":"a7c3202c-204c-4847-8af2-cb80fbce470b","resolution":{"observed_at":"2026-08-10T13:40:31.736571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1701.06538","last_updated":"2017-01-23T18:10:00Z","snapshot_observed_at":"2026-07-06T05:27:13.416519Z","submitted_at":"2017-01-23T18:10:00Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1701.06538","snapshot_observed_at":"2026-08-10T13:40:31.714813Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.714813Z"},"links":{"cited_paper":"/paper/1701.06538","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:f2295e043543425db53ba4551ff23249a14ccad3a994d0cf4f7ed385a1440369","observation_id":"75d6bb2c-7e3c-4715-9fcd-5e3d7d6ed844","resolution":{"observed_at":"2026-08-10T13:40:31.714813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.16668","last_updated":"2020-06-30T10:42:02Z","snapshot_observed_at":"2026-08-07T09:27:36.420559Z","submitted_at":"2020-06-30T10:42:02Z","title":"GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.16668","snapshot_observed_at":"2026-08-10T13:40:31.688207Z","title":"Liang, V","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.688207Z"},"links":{"cited_paper":"/paper/2006.16668","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:6d1e80079640927416aaa07cd4fe7a59f678e653725e6e3b6ba4d7b902e1b424","observation_id":"2e2c2643-cacd-4412-9d24-3376fe7ae497","resolution":{"observed_at":"2026-08-10T13:40:31.688207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.07614","last_updated":"2024-07-11T11:05:53Z","snapshot_observed_at":"2026-08-09T00:20:36.600009Z","submitted_at":"2024-07-10T12:52:49Z","title":"MARS: Mixture of Auto-Regressive Models for Fine-grained Text-to-image Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.07614","snapshot_observed_at":"2026-08-10T13:40:31.681892Z","title":"Mars: Mixture of auto-regressive models for fine-grained text-to-image syn- thesis","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.681892Z"},"links":{"cited_paper":"/paper/2407.07614","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:3686a9b02feb4e17e5b1d11233c4f1c9d0b522d54b62db4b88e99bd45e1a45a4","observation_id":"16726869-fae9-4895-8147-a60da36e5519","resolution":{"observed_at":"2026-08-10T13:40:31.681892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.03961","last_updated":"2022-06-16T20:36:07Z","snapshot_observed_at":"2026-08-10T10:13:42.613798Z","submitted_at":"2021-01-11T16:11:52Z","title":"Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.03961","snapshot_observed_at":"2026-08-10T13:40:31.668994Z","title":"Fei, Z., Fan, M., Yu, C., and Huang, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.668994Z"},"links":{"cited_paper":"/paper/2101.03961","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:1b81788337c1593fad5f4afc8f461ef6bf202ef46be57ecf24526d1cf5be5ee0","observation_id":"06815254-d7f4-444c-be73-7f0e53d1ebc6","resolution":{"observed_at":"2026-08-10T13:40:31.668994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.00396","last_updated":"2022-08-05T17:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-10-31T03:32:18Z","title":"Efficiently Modeling Long Sequences with Structured State Spaces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.00396","snapshot_observed_at":"2026-08-10T13:40:31.678996Z","title":"Efficiently modeling long sequences with structured state spaces","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.678996Z"},"links":{"cited_paper":"/paper/2111.00396","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:4117c4e2780e2e1ba92f4a730f6bfbc9a49887e0ef7d746a40f5b37547c72410","observation_id":"4f1c4b22-55f7-4c60-978a-c1dfb61d4965","resolution":{"observed_at":"2026-08-10T13:40:31.678996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02358","last_updated":"2022-05-27T15:44:26Z","snapshot_observed_at":"2026-07-06T12:05:07.980255Z","submitted_at":"2021-11-03T17:20:36Z","title":"VLMo: Unified Vision-Language Pre-Training with Mixture-of-Modality-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.02358","snapshot_observed_at":"2026-08-10T13:40:31.662404Z","title":"K., Aggarwal, K., Som, S., Piao, S., and Wei, F","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.662404Z"},"links":{"cited_paper":"/paper/2111.02358","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:7257377d20d91825c215e72ac65248f4fb8e1d0ab8547ed427f98c81e3f943ba","observation_id":"b1286cb2-0b10-49ae-8446-e159a33033df","resolution":{"observed_at":"2026-08-10T13:40:31.662404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.03120","last_updated":"2025-01-06T16:28:47Z","snapshot_observed_at":"2026-08-10T21:50:50.344907Z","submitted_at":"2025-01-06T16:28:47Z","title":"CAT: Content-Adaptive Image Tokenization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.03120","snapshot_observed_at":"2026-08-10T13:40:31.721604Z","title":"Shen, S., Yao, Z., Li, C., Darrell, T., Keutzer, K., and He, Y","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-10T13:40:31.721604Z"},"links":{"cited_paper":"/paper/2501.03120","citing_paper":"/paper/2501.16295"},"observation_digest":"sha256:3c08f91b54a17976e2d293f276c27ac22e631b6f41fe5c3fc5281d528cf23c40","observation_id":"2576ec84-5f26-41ae-942b-d852836e6f68","resolution":{"observed_at":"2026-08-10T13:40:31.721604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.16295","last_updated":"2025-01-27T18:35:05Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T21:25:18.709458Z","submitted_at":"2025-01-27T18:35:05Z","title":"Mixture-of-Mamba: Enhancing Multi-Modal State-Space Models with Modality-Aware Sparsity"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 1 inbound Pith citation observation for arXiv:2501.16295."}