{"as_of":"2026-08-22T17:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9b522ce64580f7918c4edc3084632d5e90a3b734fc05f74b702efd9e28ab32d6","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:07:38.566888Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-11T00:52:08.399401Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-11T05:05:58.687554Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"cited_work":{"arxiv_id":"2507.09394","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09394","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A random matrix theory perspective on the learning dynamics of multi-head latent attention.arXiv preprint arXiv:2507.09394","venue":null,"work_id":"0bff9c60-038f-4f00-ac0c-86cd9c1a76f0","year":null},"citing_paper":{"arxiv_id":"2605.06826","last_updated":"2026-05-07T18:28:01Z","snapshot_observed_at":"2026-08-13T02:10:14.999427Z","submitted_at":"2026-05-07T18:28:01Z","title":"How Does Attention Help? Insights from Random Matrices on Signal Recovery from Sequence Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T00:52:08.399401Z"},"links":{"cited_paper":"/paper/2507.09394","citing_paper":"/paper/2605.06826"},"observation_digest":"sha256:8211f88ff949764535186f297ec3b21abda8874c5ec04b628ad2027cc7d5c760","observation_id":"51d386c9-4ea1-4944-a112-cfc16267f99a","resolution":{"observed_at":"2026-05-11T05:05:58.694020Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.09394/citation-record","integrity":"/paper/2507.09394/integrity","json":"/paper/2507.09394/citation-record.json","paper":"/paper/2507.09394"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:41.838981Z","title":"A random matrix perspective on mix- tures of nonlinearities in high dimensions","venue":null,"work_id":"a3f4c5e9-6281-4292-bc47-99d81db34a43","year":2022},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.168580Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:cdcfd8c4e254c948c5f9ca3d97fe8b32b6aa11e693d49f649a6acafe371094cd","observation_id":"ee683523-97b9-4efe-a703-6c44ed8cdb82","resolution":{"observed_at":"2026-08-06T18:07:41.900833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:41.632560Z","title":"Self-attention networks localize when QK- eigenspectrum concentrates","venue":null,"work_id":"fc1268eb-06c2-4aa3-ace7-3de0c7f339a0","year":2024},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.209065Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:8334a881d982e191eb1d4f5b31abf277d07e6beb64397a449409efb1b84a5b40","observation_id":"8cf650c4-73bd-4dc8-86ba-fab5670b39a9","resolution":{"observed_at":"2026-08-06T18:07:41.736678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:41.467365Z","title":"Random matrix theory improved fr ´echet mean of symmetric positive definite matrices","venue":null,"work_id":"192937be-f28b-4ec1-8031-88d941f66099","year":2024},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.253547Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:39cf3927fdecb8d578c979d0de53b51bb5996400774378224af61e1756fdd0ac","observation_id":"3ee33393-c328-446e-824b-1de74fadeb71","resolution":{"observed_at":"2026-08-06T18:07:41.568076Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:41.190446Z","title":"A random matrix ap- proach to echo-state neural networks","venue":null,"work_id":"8609aaac-6a7c-4f65-a242-51ed2696d860","year":2016},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.296250Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:da8ada4e92834705430166fabbef79c6148bb603e79cc6a94aad4585f91d96fe","observation_id":"349b43ec-d02e-4ff6-83fb-abf279ddd3c0","resolution":{"observed_at":"2026-08-06T18:07:41.319175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:41.018648Z","title":"A random matrix theory perspective on the spectrum of learned features and asymptotic generalization capabilities","venue":null,"work_id":"c532459f-3723-489e-891d-567099bbe857","year":2025},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.345132Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:f2329006e924ae66945ad88774ba53e5237fb09a44043993d00d72627faa60a9","observation_id":"81c403de-6cd6-47f2-a51c-71fb29dd1ce1","resolution":{"observed_at":"2026-08-06T18:07:41.118098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:40.860333Z","title":"Random matrix analysis to balance between supervised and unsupervised learning under the low density separation assumption","venue":null,"work_id":"5a14d34b-9d2b-4dd1-81e8-03260d6f2f67","year":2023},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.408099Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:629fdcec3ec1bce8a8b4632bf737a0db9830f1a948c0c569f2fbe5f105018872","observation_id":"a11f441f-56f7-4e0d-9000-0ac68f94eb44","resolution":{"observed_at":"2026-08-06T18:07:40.943455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:40.622029Z","title":"Maximizing the potential of synthetic data: Insights from ran- dom matrix theory","venue":null,"work_id":"15f973ce-42e7-46ca-962f-bba0307b2671","year":2025},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.472843Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:9268c56b59060a508fcfe1c2248919e276532e14547385a7d9750ed8a0964a9d","observation_id":"98dffd79-1563-419d-923b-573a5551b30b","resolution":{"observed_at":"2026-08-06T18:07:40.719825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:40.422064Z","title":"Analysing multi-task regression via random matrix theory with application to time series forecasting","venue":null,"work_id":"f58f5de2-742b-45c7-a5c9-17c7e4de774e","year":2024},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.534906Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:d876b0b157f9e4f2888865a8e1ddbad80f87d7820de49b1ce03303ca7550a5ef","observation_id":"45bcfe2a-a32e-45dd-8d56-451575538272","resolution":{"observed_at":"2026-08-06T18:07:40.541879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14975","last_updated":"2024-04-05T10:45:19Z","snapshot_observed_at":"2026-08-19T07:00:26.926239Z","submitted_at":"2023-06-26T18:01:47Z","title":"The Underlying Scaling Laws and Universal Statistical Structure of Complex Datasets","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14975","snapshot_observed_at":"2026-08-06T18:07:37.592873Z","title":"The underlying scaling laws and universal statistical structure of complex datasets","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.592873Z"},"links":{"cited_paper":"/paper/2306.14975","citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:5d7684f5ddfadaa85cae0b73a1a346f0f8016369ea96e8e4b0425267c8d89d0e","observation_id":"25ea31cf-a862-4943-97aa-7e3c08b5dcae","resolution":{"observed_at":"2026-08-06T18:07:37.592873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:40.255975Z","title":"Mix-LN: Unleashing the power of deeper layers by combining pre-LN and post-LN","venue":null,"work_id":"d1908254-758e-48d6-b31c-4fdc853fa5c1","year":2025},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.662358Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:28ad912a2553f24d1aaa6704575f47cfe5d8925507b3572a80bebb0777b221a1","observation_id":"8971b61d-d2b0-4d5a-b90d-7d9437f98b6b","resolution":{"observed_at":"2026-08-06T18:07:40.349433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:40.083583Z","title":"The dynamics of learning: A random matrix approach","venue":null,"work_id":"84ea3d19-8925-41a6-8d98-07897804a966","year":2018},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.723896Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:645d28717de6255523c8bbfc0dbcc2aba54191ea576c84b380c1b3556af78c6a","observation_id":"c6ac8cd5-f5f1-4af1-8e41-5c10500bd25b","resolution":{"observed_at":"2026-08-06T18:07:40.148126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04434","last_updated":"2024-06-19T06:04:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-07T15:56:43Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04434","snapshot_observed_at":"2026-08-06T18:07:37.781457Z","title":"Deepseek-v2: A strong, economical, and efficient mixture-of-experts language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.781457Z"},"links":{"cited_paper":"/paper/2405.04434","citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:87a960d0d4b5332ca76456ece809309821222b002cebc643028d928761b2e141","observation_id":"2b056b2e-0d9d-42bc-877e-fa12dc11cac8","resolution":{"observed_at":"2026-08-06T18:07:37.781457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-18T18:18:37.449517Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T18:07:37.848948Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.848948Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:e6138926881650c87180392623646754fc2459904605813426d3d1362aac6d15","observation_id":"8c6a1f5a-6fe6-46b8-be9a-d5c50b87b7b5","resolution":{"observed_at":"2026-08-06T18:07:37.848948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:39.922976Z","title":"Distribution of eigenvalues for some sets of random matrices","venue":null,"work_id":"66fcd489-7216-4f2e-a7ed-bc26e98c4052","year":1967},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.905643Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:ae69360b872596b88e957c86fed9c4a7c2a8e1a4fd175d30a0e5a92d703fe98a","observation_id":"230f14ec-31ac-4344-95e8-7acaf51b5acc","resolution":{"observed_at":"2026-08-06T18:07:40.021515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:39.715378Z","title":"Implicit self-regularization in deep neural net- works: Evidence from random matrix theory and implications for learning","venue":null,"work_id":"d874b21e-981a-4f4a-904c-d21730c09f73","year":2021},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:37.956487Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:f06f4e9905694079bd5dc083dad83a786cd62dc750a176c6a7c7b3c5cd449ff7","observation_id":"43395909-49b0-4b56-84a4-8915f5ab0f54","resolution":{"observed_at":"2026-08-06T18:07:39.837235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07864","last_updated":"2025-06-12T11:45:57Z","snapshot_observed_at":"2026-08-10T11:58:20.177769Z","submitted_at":"2025-02-11T18:20:18Z","title":"TransMLA: Multi-Head Latent Attention Is All You Need","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.07864","snapshot_observed_at":"2026-08-06T18:07:38.019228Z","title":"Transmla: Multi-head latent attention is all you need","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.019228Z"},"links":{"cited_paper":"/paper/2502.07864","citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:8795c8d9d1c8e2fca41dc5727ce187b21260cbb3f50160621d1a9e88ed6fd625","observation_id":"db4c0ef7-315a-436c-89d6-569d992d4d08","resolution":{"observed_at":"2026-08-06T18:07:38.019228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:39.490804Z","title":"Geometry of neural network loss surfaces via random matrix theory","venue":null,"work_id":"05a6a6ca-c65d-4e50-b13f-020321a6ee1a","year":2017},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.104880Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:41afd1b4c23e6a0c5b081b1e4e5be35fdfbf9b5d2301c6b04e76f1eef03bc48d","observation_id":"cb10a6e6-f7da-4471-82ed-b3178aad6735","resolution":{"observed_at":"2026-08-06T18:07:39.600408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:38.183231Z","title":"Nonlinear random matrix theory for deep learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.183231Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:eca08df1bfd3b65568098d17b4bc3d6f5accfb6953b84015fe22c968e540baf6","observation_id":"f8ca5be1-61db-472c-a785-0e266df756c9","resolution":{"observed_at":"2026-08-06T18:07:38.183231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:38.258595Z","title":"Locating information in large language models via random matrix theory","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.258595Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:5840d4c86605fa22f5d1fa13e4dfa2dd866eba764e528c1208b89ada6a25ca1a","observation_id":"2aa809c4-6732-48d9-af69-59694030d5fe","resolution":{"observed_at":"2026-08-06T18:07:38.258595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:39.310582Z","title":"Random matrix theory analysis of neural network weight matrices","venue":null,"work_id":"7a11cc02-7e41-4eaf-b5b0-276204f0abd9","year":2024},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.333559Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:cda2643803500109bcbdffe1f1b8b99a1e938b96c28da56b80edef234c6e0b1b","observation_id":"f33cf979-feb5-402e-82c3-4c4caf4073d4","resolution":{"observed_at":"2026-08-06T18:07:39.412438Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:39.119842Z","title":"Random ma- trix improved covariance estimation for a large class of metrics","venue":null,"work_id":"da3a16fb-e19f-48d5-9de7-ab19677ebac5","year":2019},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.399805Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:37442b7c497d213eb67b04e7dd0d64e2d8ce3cf21fd84377d92988d8d76bc6e2","observation_id":"36013000-9d1b-44b9-aaea-1875b88768bc","resolution":{"observed_at":"2026-08-06T18:07:39.199843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:38.938093Z","title":"More than a toy: Random matrix models pre- dict how real-world neural representations generalize","venue":null,"work_id":"e3e8b31c-1b89-498a-94a5-3795dc69b9b4","year":2022},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.453182Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:02dc801318459827b92d5c3bd8d240289bd7afbc83b5c5ec33a47d3f200c5959","observation_id":"5cd7fe0c-24b6-48c5-b213-3872f97779b1","resolution":{"observed_at":"2026-08-06T18:07:39.022322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:07:38.566888Z","title":"Insights into deepseek-v3: Scaling challenges and reflections on hardware for ai architectures","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T18:07:38.566888Z"},"links":{"citing_paper":"/paper/2507.09394"},"observation_digest":"sha256:d8e8951eb7dea9e0257de3c0da0f4c67d5773490a1e98908c7190124075f0f64","observation_id":"4d42d40f-dc65-4ff9-b0d8-1b5136db0d19","resolution":{"observed_at":"2026-08-06T18:07:38.566888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.09394","last_updated":"2025-07-12T20:31:07Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T03:34:09.430242Z","submitted_at":"2025-07-12T20:31:07Z","title":"A Random Matrix Theory Perspective on the Learning Dynamics of Multi-head Latent Attention"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":0,"verified_fuzzy":16},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 1 inbound Pith citation observation for arXiv:2507.09394."}