{"as_of":"2026-08-23T17:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2f9069c969413f1e3e7e5af235bf6657cebba861a3a9f47ec0fcd206e64ce48b","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-21T05:01:27.451161Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.21486/citation-record","integrity":"/paper/2605.21486/integrity","json":"/paper/2605.21486/citation-record.json","paper":"/paper/2605.21486"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.10684","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T17:20:00.901635Z","title":"arXiv preprint arXiv:2601.10684 , year =","venue":null,"work_id":"13f1aa21-8e86-417e-aac4-431842500cb9","year":2026},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:3874e20bd3c2696d3b0d18e685edfad989666afc018be51fac20442cc494cc0d","observation_id":"586a96da-6d6d-4d44-9db4-2a79ab78f9e5","resolution":{"observed_at":"2026-05-21T05:03:57.948549Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Power lines: Scaling laws for weight decay and batch size in LLM pre-training","venue":null,"work_id":"f4a23843-0d68-4aec-9851-b4ba04e0a3af","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:0662f47b14c922ee102917d93cc984291094437fadae5c313dd3ed7b4ebb08bb","observation_id":"a1d05a41-616c-404b-b64e-2b876bfdddbe","resolution":{"observed_at":"2026-05-21T05:03:58.660494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling optimal LR across token horizons","venue":null,"work_id":"785f82d4-0bde-460a-9399-81753c21f80f","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:0c4ab8758886b1ba34365784a3ae65a7fc3c0576614041d31dfd539f9b539ef6","observation_id":"7a263e0a-64e2-41d2-be0f-f3adc7563951","resolution":{"observed_at":"2026-05-21T05:03:58.643714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-consistent dynamical field theory of kernel evolution in wide neural networks","venue":null,"work_id":"2e7e8db2-db2b-46a9-83c9-186630ba9b47","year":2022},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:70dc59c22d621818579002de8d6b91f42daa2dfd89f101277d13583450865590","observation_id":"ed4ce618-d317-494a-b97d-f3298cf78f88","resolution":{"observed_at":"2026-05-21T05:03:58.640010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Infinite limits of multi-head transformer dynamics","venue":null,"work_id":"60d2974d-880f-4dd0-92ba-f05f80f68abc","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:d52eb8edcef5480a8daad0c5a7b00cc4ab1078d630fb8fa8f9469056a07f4c42","observation_id":"1c25a29d-c012-404b-b05f-571ab0cc1c65","resolution":{"observed_at":"2026-05-21T05:03:58.662895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Depthwise hyperparam- eter transfer in residual networks: Dynamics and scaling limit","venue":null,"work_id":"b75f5bbc-e7fd-496c-9dbd-a3b16f7dcd38","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:4e222191324c8599e4dc208ff829627824d6e7b40e43226b4e5461ebc45ea00a","observation_id":"f037e25d-30d7-42de-ae5b-2b44120b27e0","resolution":{"observed_at":"2026-05-21T05:03:58.670796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.02954","last_updated":"2024-01-05T18:59:13Z","snapshot_observed_at":"2026-08-14T19:47:04.330126Z","submitted_at":"2024-01-05T18:59:13Z","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","version":1},"cited_work":{"arxiv_id":"2401.02954","doi":"10.48550/arxiv.2401.02954","metadata_source":"pith","pith_arxiv_id":"2401.02954","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","venue":"cs.CL","work_id":"01b10587-025b-499d-8ba3-7a538d24c2d6","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2401.02954","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:bd753f662a10b71363fb363612f28ac0ead550d80043bb4d0ba637ebcb7b2c08","observation_id":"302c5891-7b69-4413-9158-6fdef40f214b","resolution":{"observed_at":"2026-05-21T05:03:57.951136Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Don’t be lazy: Completep enables compute-efficient deep transformers","venue":null,"work_id":"0b356d42-71fc-437b-b8f0-0a6fa5d99966","year":null},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:422586eaa6e8e8aab37e65831556af089b65ea526b19f8fe2e61515cea8cfc44","observation_id":"2d7c648a-5696-49b8-b54b-b269cf976d9f","resolution":{"observed_at":"2026-05-21T05:03:58.668839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.01618","doi":"10.48550/arxiv.2505.01618","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Don’t be lazy: Completep enables compute-efficient deep transformers","venue":"ArXiv.org","work_id":"85f11780-ed20-4881-8d31-bb0834b58027","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:d4928c93f5d68130ebf5d85e4c2ebf96b51f0dc03a94ced63992f6e42c71bc0c","observation_id":"1d32c1bb-2139-42dc-8f2d-fa80a1568b2d","resolution":{"observed_at":"2026-05-21T05:03:57.968125Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sparse maximal update parameterization: A holistic approach to sparse training dynamics","venue":null,"work_id":"6bbf6996-0ca5-431a-916a-5156823ad519","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:b3a460ea95f970638bb42a2758edf6f836e75f263cb75c32734e753e6396be42","observation_id":"e5b9bc47-bec3-4567-86ba-5d3fa806b059","resolution":{"observed_at":"2026-05-21T05:03:58.674209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling exponents across parameterizations and optimizers","venue":null,"work_id":"3344563f-c73f-47fb-8f76-1fe987316e63","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:b4d49b3e9a766031944b484cb99e395346b03ebb3024b47165a5c409fcae28c5","observation_id":"26446428-f3be-4386-8359-d849220f101f","resolution":{"observed_at":"2026-05-21T05:03:58.676694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.22768","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Understanding the mechanisms of fast hyperpa- rameter transfer","venue":null,"work_id":"8057482a-2cb6-42e6-a46f-e6542c181011","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:21edb323ab391f0ab6973eff056ccfcacf61460e4589a9428736eb22dcd38eae","observation_id":"f0f791cf-69e3-41cb-9a38-d657860c477a","resolution":{"observed_at":"2026-05-21T05:03:57.959670Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A loss curvature perspective on training instabilities of deep learning models","venue":null,"work_id":"97a1751a-bd51-4038-a337-3f7f676df78d","year":2022},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:0950272ccfd77656da76454f6a75f71b2ceb924be7e69e69d805e016af1440b1","observation_id":"b19202a0-b40d-4f11-bef0-9735ce7c424d","resolution":{"observed_at":"2026-05-21T05:03:58.656727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"$\\boldsymbol{\\mu}\\mathbf{P^2}$: Effective sharpness aware minimization requires layerwise perturbation scaling","venue":null,"work_id":"fc257aed-03d6-4607-a80c-a0f33d22c23e","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:0576d5a9f8425a5a20589c187ee0c98ae676adc8cbcda210c98e66e46604f4fc","observation_id":"4f7d4861-cf7c-402e-a6ab-c38b0e9d9ef5","resolution":{"observed_at":"2026-05-21T05:03:58.652917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A proof of learning rate transfer under $\\mu$p","venue":null,"work_id":"55891953-e0fb-48cc-89ce-ae948d23cf87","year":2026},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:54fe0e442fc4b0d4c0752b6bc2dd55152532f9ed3c091e677423ae176a55f624","observation_id":"ceeecce6-f2dc-4c8c-8262-1b0c501f6b54","resolution":{"observed_at":"2026-05-21T05:03:58.654977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Optimal embedding learning rate in llms: The effect of vocabulary size","venue":null,"work_id":"43f61303-1334-4e2c-b1a4-4457fdc4e2ae","year":null},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:41f094f3c4cb571926d398cb8d47a5d173520eab27f80d6d883596fb464994e8","observation_id":"73b515ef-7989-41b6-931b-51d9f8343192","resolution":{"observed_at":"2026-05-21T05:03:58.651223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15025","last_updated":"2025-06-17T23:57:30Z","snapshot_observed_at":"2026-08-18T09:46:27.882306Z","submitted_at":"2025-06-17T23:57:30Z","title":"Optimal Embedding Learning Rate in LLMs: The Effect of Vocabulary Size","version":1},"cited_work":{"arxiv_id":"2506.15025","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15025","snapshot_observed_at":"2026-06-30T17:34:57.593618Z","title":"and Liu, L","venue":null,"work_id":"0c9aed32-c6c5-4249-a693-b94ad28fb680","year":null},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2506.15025","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:7ecb049e25fcd791444d3e3eec17963de0e2a9eae27bcc6d6ee1cb6cd95a1e94","observation_id":"b12fee4f-c0fe-4629-9cfd-5c377a29cff1","resolution":{"observed_at":"2026-05-21T05:03:57.956920Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning to grok: Emergence of in-context learning and skill composition in modular arithmetic tasks","venue":null,"work_id":"1fd59379-2985-4f48-a9f7-f8ad2e006a30","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:0ca37efee1c300288e6777521ae117c49c73c22ad1b7bb266fd3df8f3b2f2b45","observation_id":"2a223616-955e-4695-beef-6ed93ad0b5d8","resolution":{"observed_at":"2026-05-21T05:03:58.647551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An empirical analysis of compute-optimal large language model training","venue":null,"work_id":"4daf66b0-6a60-4c58-940e-645ca11e9256","year":2022},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:3b4962082d2bb75d1d77cd76c902e9628bbdeec3a54b87818cf7b6936777be0c","observation_id":"341c8c77-a45e-43c5-b927-0d000c51a526","resolution":{"observed_at":"2026-05-21T05:03:58.649109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MiniCPM: Unveiling the potential of small language models with scalable training strategies","venue":null,"work_id":"9c501f39-bcff-4e51-ac25-722214c7423d","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:dea5eb3708c529fd549879f2c41765ea58e4ada139e99cec660b850928b3dd44","observation_id":"a25c7cef-17ab-492a-8c89-fa20506e0592","resolution":{"observed_at":"2026-05-21T05:03:58.658846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.20205","last_updated":"2026-05-21T16:25:08Z","snapshot_observed_at":"2026-08-17T09:52:56.757036Z","submitted_at":"2026-01-28T03:02:30Z","title":"Hyperparameter Transfer with Mixture-of-Expert Layers","version":3},"cited_work":{"arxiv_id":"2601.20205","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.20205","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hyperparameter Transfer with Mixture-of-Expert Layers","venue":"cs.LG","work_id":"a17948d6-cdb2-4dbb-890a-1fc24662c6d5","year":2026},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2601.20205","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:2fe374a5fb2e199abe88891eb553eb20b5d72bb953f5b653ba1d18888a41c4ef","observation_id":"105ca246-4592-4698-8265-beb37beba02a","resolution":{"observed_at":"2026-05-22T03:04:27.931887Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Muon: An optimizer for hidden layers in neural networks","venue":null,"work_id":"09a77098-a8e7-4dda-8bc0-44bac68cf003","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:fe1e5e8b2c34b2c14e8e3ef1e3c27717a92cd2e24a1f5790b49bc239f4f3e633","observation_id":"5df69093-255e-45e3-90fa-e2d1e9061e4e","resolution":{"observed_at":"2026-05-21T05:03:58.664825Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Why warmup the learning rate? underlying mechanisms and improvements","venue":null,"work_id":"287146d4-5aee-4229-81bf-c67b7976d523","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:017bcfb2d528dc5108adcb9d216861774bca306d8925fe4ee454f0942e82f2c8","observation_id":"08d9609b-8193-47fc-b997-1d026b4fcc94","resolution":{"observed_at":"2026-05-21T05:03:58.638192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Universal sharpness dynamics in neural network training: Fixed point analysis, edge of stability, and route to chaos","venue":null,"work_id":"96459755-cb75-47e0-8bb4-4ec63bd3b754","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:ea188928b3f1207d48f0a585a341d952bc5f4575514cd7c32b8acf6034a38c95","observation_id":"c3c2da91-5055-4caf-adbc-6747afaba377","resolution":{"observed_at":"2026-05-21T05:03:58.641688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-08-13T17:41:53.092611Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":"2001.08361","doi":"10.1145/3616855.3635845","metadata_source":"pith","pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling Laws for Neural Language Models","venue":"cs.LG","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","year":2020},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:d365d567218690b02f57fb6362c8e43d7d1abb905805c97afeeb47e200388f75","observation_id":"1f7aefa1-9268-4b7b-8ec0-b949486be19b","resolution":{"observed_at":"2026-05-21T05:03:57.953743Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.15029","last_updated":"2026-06-29T19:57:33Z","snapshot_observed_at":"2026-08-17T20:18:15.051321Z","submitted_at":"2026-02-16T18:59:55Z","title":"Symmetry in language statistics shapes the geometry of model representations","version":3},"cited_work":{"arxiv_id":"2602.15029","doi":"10.48550/arxiv.2602.15029","metadata_source":"pith","pith_arxiv_id":"2602.15029","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Symmetry in language statistics shapes the geometry of model representations","venue":"cs.LG","work_id":"d7f0fbcb-6a3a-4efc-ad83-e97bf6e48804","year":2026},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2602.15029","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:2b985bf117cb4bbe0587df1c2f222304764d03ef390a76daa20ade03301179ad","observation_id":"f9d77f6e-8539-4e61-83d1-6ba760fdc530","resolution":{"observed_at":"2026-07-01T01:17:13.480674Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-05-23T11:22:42.537198+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T11:22:42.537198+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"nanoGPT: The simplest, fastest repository for training/finetuning medium-sized gpts","venue":null,"work_id":"8a40833f-eade-43ce-b4e5-e45390ff3027","year":2022},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:3eb9b73ad0cd6cc6d55d3d386a4852b057382398af6c92ebb8c7e52a68318807","observation_id":"36b6ff84-5963-4b3d-b089-7dfdddbee848","resolution":{"observed_at":"2026-05-21T05:03:58.666566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Weight decay may matter more than µp for learning rate transfer in practice","venue":null,"work_id":"7bcb8f2f-be2d-43bd-8c94-31e4ce6b6e1c","year":2026},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:a548aeea6429ed87c7da5f5cd02d5239222e099a5eed8f37c4cdbd58615d3db6","observation_id":"4dd91632-60e5-431d-a328-4785cbf89c4f","resolution":{"observed_at":"2026-05-21T05:03:58.636273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cifar-100 (canadian institute for advanced research)","venue":null,"work_id":"e3341fa0-de3b-496b-b0c0-f3b23a8cb6e2","year":null},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:7ecadda1ade4dd5020cc0a922d4e6fd31cf1a5f6e6210b39279f1f7c375db367","observation_id":"68afa20d-3f98-4b2e-9507-7f8edb63c538","resolution":{"observed_at":"2026-05-21T05:03:58.631949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04715","last_updated":"2025-08-19T09:45:25Z","snapshot_observed_at":"2026-08-18T04:12:29.569605Z","submitted_at":"2025-03-06T18:58:29Z","title":"Predictable Scale: Part I, Step Law -- Optimal Hyperparameter Scaling Law in Large Language Model Pretraining","version":7},"cited_work":{"arxiv_id":"2503.04715","doi":"10.48550/arxiv.2503.04715","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.04715","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Predictable scale: Part i–optimal hyperparameter scaling law in large language model pretraining","venue":"ArXiv.org","work_id":"dc2607d8-8c5c-41ec-a2af-c022c7c4d0ee","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2503.04715","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:94f7d3340ce5d059a758b0a2645f13f1d02a98739f3200ada30fb8d7f310151e","observation_id":"931597dd-a8b5-4142-8954-839468e0a7bc","resolution":{"observed_at":"2026-05-21T05:03:57.973859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Adaptive optimization in the $\\infty$-width limit","venue":null,"work_id":"bb44afa1-c0c0-4829-b22a-bee904b1ae12","year":2023},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:7a74343b30e4f1160ab575b3d6dce2d58e929a6ad07ea9b94222d6f21a38355c","observation_id":"aa9a5e87-a9a4-4286-a65d-e1b5177d18c6","resolution":{"observed_at":"2026-05-21T05:03:58.684429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:e9c5f4b3de9903e842f0ae92003371d51c4cc8f85b603fb251a8d3f5609d0b52","observation_id":"40254e0d-da91-4b13-b354-1d44a5b8f043","resolution":{"observed_at":"2026-05-21T05:03:57.976267Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Decoupled weight decay regularization","venue":null,"work_id":"a521b5a6-ed0b-4825-aca1-d79466420614","year":2019},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:7f079db1f8a0b6019583468ec77676187ee58fd07c1389f74d7d5ca70e858f51","observation_id":"e86398ba-6468-4586-9b1a-3c1329e20761","resolution":{"observed_at":"2026-05-21T05:03:58.630063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.09752","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T00:09:15.185738Z","title":"µ-parametrization for mixture of experts","venue":null,"work_id":"62e853f3-89d4-4c66-b43d-a2a6c7f0db29","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:e99b87952897d362fe9398094590bca968b6277fd4b659c05de9735ff8bd0003","observation_id":"4e04130c-406b-4290-ada1-1fdc75036a62","resolution":{"observed_at":"2026-05-21T05:03:57.965196Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Progress measures for grokking via mechanistic interpretability","venue":null,"work_id":"30908d7e-7d8c-4c4a-9170-4052465161ed","year":2023},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:664f81b46647076f45b189aa82212ad4da065fd760d00f02ea10e88be64a2ef9","observation_id":"fd7c740f-8eb3-4287-9736-607561d38a74","resolution":{"observed_at":"2026-05-21T05:03:58.625687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Super consistency of neural network landscapes and learning rate transfer","venue":null,"work_id":"c0ca4fe8-1fef-46f2-a726-5f8f713d1ff0","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:0dd768f696716ab6e3c952d4cc6e9a2ce3320080d562a078d49a6ffea0dcc889","observation_id":"81cf0dfb-a06a-45a5-a40c-d8187920c6b4","resolution":{"observed_at":"2026-05-21T05:03:58.627557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00656","last_updated":"2025-10-08T07:50:45Z","snapshot_observed_at":"2026-08-17T13:58:40.683829Z","submitted_at":"2024-12-31T21:55:10Z","title":"2 OLMo 2 Furious","version":3},"cited_work":{"arxiv_id":"2501.00656","doi":"10.48550/arxiv.2501.00656","metadata_source":"pith","pith_arxiv_id":"2501.00656","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2 OLMo 2 Furious","venue":"cs.CL","work_id":"9ef0dc2b-fdfe-4f14-b235-ef7556dc709a","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2501.00656","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:d92dfc7e790cac81e792e6b9d24574e226ff58aba5970926cfb56266bf31d222","observation_id":"f3969d39-71f5-4c4a-946f-e67e31b4288c","resolution":{"observed_at":"2026-05-21T05:03:57.978819Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-11T05:19:24.20254+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T05:19:24.20254+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The fineweb datasets: Decanting the web for the finest text data at scale","venue":null,"work_id":"7f5043db-ac3e-4037-88ad-40172c52d4d4","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:721edf19c9263eb11a9fc2a27552f6d07c42c87521c29169e0de7b6591ff626a","observation_id":"b71bc578-9523-445a-9ada-6d9114cfa116","resolution":{"observed_at":"2026-05-21T05:03:58.634492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Resolving discrepan- cies in compute-optimal scaling of language models","venue":null,"work_id":"347565ad-0278-4ebb-bea4-3aea9ea4524d","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:7dbb1abbf1d4f3445e7943ec0d6a9a178db1c8dc53525f1b578d1ed944053fbe","observation_id":"2591375f-72b6-434c-b7b8-ded6f5501d98","resolution":{"observed_at":"2026-05-21T05:03:58.645280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.05620","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T12:06:55.443959Z","title":"Hyperparameter transfer enables consistent gains of matrix-preconditioned optimizers across scales","venue":null,"work_id":"046aa1a9-fd5f-4b43-ad97-505a74d192ca","year":2026},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:96cc3a3b0f4e82241f8755788519a4c0a72e3cb29f8b94280e8b4692a8b5e7f1","observation_id":"f090bda2-61a0-4e8b-a103-c053707c5cbf","resolution":{"observed_at":"2026-05-21T05:03:57.942472Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-08-17T18:50:07.059564Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.15115","doi":"10.1145/3581783.3612503","metadata_source":"pith","pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5 Technical Report","venue":"cs.CL","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:4a0ee2fe9be7c12ff6cc6c6eecfc4396523f05c2bfb1c4808082851830c7d51d","observation_id":"8d1391f0-43b8-4764-9e95-337cef06a1a9","resolution":{"observed_at":"2026-05-21T05:03:57.981696Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models are unsupervised multitask learners","venue":null,"work_id":"edd00bce-ce05-4592-8de5-248e16b58e39","year":2019},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:ab8a5f2ad848c17c7bc3e2570c73d36a2fe974bedf070f06e2be97ebbc2d79a1","observation_id":"12cab608-8552-497d-8cdc-825ae6c0ae81","resolution":{"observed_at":"2026-05-21T05:03:58.691495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Roberts, Sho Yaida, and Boris Hanin.Frontmatter, page i–iv","venue":null,"work_id":"f97ffb53-cd50-4bf8-be98-f58dd6dd882f","year":2022},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:90878b8883a90d1343adeb2194536b06b09c1323bcb67ed9fd30722f0222e29f","observation_id":"70b76eaa-2f5f-4d4a-ae97-b91841feaf8b","resolution":{"observed_at":"2026-05-21T05:03:58.693583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.07301","last_updated":"2020-04-18T21:06:06Z","snapshot_observed_at":"2026-08-19T01:28:53.957925Z","submitted_at":"2020-01-21T01:02:21Z","title":"On the infinite width limit of neural networks with a standard parameterization","version":3},"cited_work":{"arxiv_id":"2001.07301","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2001.07301","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"On the infinite width limit of neural networks with a standard parameterization","venue":null,"work_id":"1a6c074f-9ed5-41b0-a3a9-c26db565857b","year":2001},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2001.07301","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:f61204054ed6376c1208cc71452e6e3971778097342f5489d4cdb630a0b785ba","observation_id":"437a4816-6dc1-453d-894b-da3670683c40","resolution":{"observed_at":"2026-05-21T05:03:57.984792Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"(how) can transformers predict pseudo-random numbers? InForty-second International Conference on Machine Learning","venue":null,"work_id":"1de57395-786f-4f9d-b1a8-2491500cfc50","year":2025},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:164d15587cc94d993befe490e15f0ffc141b491d5fa362eaaf8fc4f3ef70eb98","observation_id":"c5221330-e3ca-49df-afc5-15af3e80a49a","resolution":{"observed_at":"2026-05-21T05:03:58.685637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"On feature learning in structured state space models","venue":null,"work_id":"f72c5441-2eb3-4bab-bc50-a13302393b00","year":null},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:5ce8839222d0a761b0b71f035d6ca955f2de5697e0a8fef8e3bed1d5d27248a8","observation_id":"ae3590ca-3896-46be-9f10-74e6a61f9fd2","resolution":{"observed_at":"2026-05-21T05:03:58.688013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c7d5698d-7ff6-4e91-8e70-a98ef2ce124f","year":null},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:3aee123b929b83d40a5369166f3611c7fe2567d78ae2d01abab5c59b6eb52540","observation_id":"01963f8e-bf7f-4d72-98e7-4afd9d0a0ac6","resolution":{"observed_at":"2026-05-21T05:03:58.689757Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Attention is all you need","venue":null,"work_id":"7631e965-b69a-45f6-aa91-1bb51d7752aa","year":2017},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:8905ecdc56a78b5c1ac79fc9bc5afb7a87f872eb6b0ff3c56fc2ccd3097f0148","observation_id":"9080fe0e-3163-4552-ad8d-c1cf76545eda","resolution":{"observed_at":"2026-05-21T05:03:58.695277Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.04909","last_updated":"2022-10-18T17:59:59Z","snapshot_observed_at":"2026-08-16T16:25:31.157477Z","submitted_at":"2022-10-10T18:00:01Z","title":"Meta-Principled Family of Hyperparameter Scaling Strategies","version":2},"cited_work":{"arxiv_id":"2210.04909","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2210.04909","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Meta-principled family of hyperparameter scaling strategies","venue":null,"work_id":"010456cb-6a59-4997-866f-92252805a1c2","year":2022},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"cited_paper":"/paper/2210.04909","citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:6794ce818a724985517c2623c0b8eb2e502b6816c71d79cf1e97f2bfd6d50813","observation_id":"bcc20965-99b7-4629-a9b1-c38ce91e6e7b","resolution":{"observed_at":"2026-05-21T05:03:57.970923Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c36a47a5-4efa-4b55-9bd3-ae2159a268f6","year":2021},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:6c5a80c9eddce2dbed36339291c6795e52ba0dd3033cf200aebec3ec503850cb","observation_id":"a8438573-213e-4618-8cbf-e6e9936bf45a","resolution":{"observed_at":"2026-05-21T05:03:58.681913Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tuning large neural networks via zero-shot hyperparameter transfer","venue":null,"work_id":"1654c864-4a04-403e-a16c-7f0ca7b3ab92","year":2021},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:7e89ea929cbeb4510b1ba7103a5245c48a4639126b119d32e14983a85592d63b","observation_id":"9ab1a6ab-dfb7-43b5-801b-e8df0e228b85","resolution":{"observed_at":"2026-05-21T05:03:58.679500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tensor programs VI: Feature learning in infinite depth neural networks","venue":null,"work_id":"f87e9615-9c9e-48d9-acc2-c0978e60b0c6","year":2024},"citing_paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-21T05:01:27.451161Z"},"links":{"citing_paper":"/paper/2605.21486"},"observation_digest":"sha256:29a8e820f407baf741b5aed2dad4077ad448adc9504f382a9ac513ee0581eb66","observation_id":"61b9a4db-fcf8-4e77-b4bc-1e8260a5bd0d","resolution":{"observed_at":"2026-05-21T05:03:58.623087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.21486","last_updated":"2026-05-20T17:59:40Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T09:06:52.127366Z","submitted_at":"2026-05-20T17:59:40Z","title":"Quantifying Hyperparameter Transfer and the Importance of Embedding Layer Learning Rate"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":2,"verified_exact":13,"verified_fuzzy":34},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 0 inbound Pith citation observations for arXiv:2605.21486."}