{"as_of":"2026-08-08T02:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:95a09eba56a58fba34aa9a6e16f6d2c77f409c68c332d53c73e40b64fa04ce0c","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:14:20.461704Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.22389/citation-record","integrity":"/paper/2506.22389/integrity","json":"/paper/2506.22389/citation-record.json","paper":"/paper/2506.22389"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T22:14:18.370267Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.370267Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:b9ee985ea12f4c729b2c7dfd676a183282dbb26595b366dd33d69043f40e2034","observation_id":"b0975920-8ccb-4fc5-bfeb-2264c07accdc","resolution":{"observed_at":"2026-08-06T22:14:18.370267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:18.796382Z","title":"Jia Deng, Wei Dong, Richard Socher, Li-Jia Li, Kai Li, and Li Fei-Fei","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.796382Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:3946e2a872df0b7e788d46905f592fe6f9e6b331adb280a532d9aef48d32f90e","observation_id":"2482fa62-7024-41dd-992c-ebe0d7109451","resolution":{"observed_at":"2026-08-06T22:14:18.796382Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05171","last_updated":"2025-02-17T17:14:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-07T18:55:02Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05171","snapshot_observed_at":"2026-08-06T22:14:18.944144Z","title":"Scaling up test-time compute with latent reasoning: A recurrent depth approach","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.944144Z"},"links":{"cited_paper":"/paper/2502.05171","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:c7762bb3e1511debf9ca72b8863d2394a0a250ca6d2eeb3410440f1c9f97c78a","observation_id":"280f67af-b0b0-4421-a7a3-b66c0271f621","resolution":{"observed_at":"2026-08-06T22:14:18.944144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.498186Z","title":"This may be because we are not considering a setting with high sparsity","venue":null,"work_id":"638c18dd-208a-47d2-aae9-95280f3cb351","year":2017},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:20.148222Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:ec81d10c467d7692c6d20b0812f8b7ec9ffc94cd0d1778960f2492a43040caae","observation_id":"1ca5aea9-79ea-4d95-a551-6c355c8775e5","resolution":{"observed_at":"2026-08-06T22:14:21.567025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T22:14:19.357714Z","title":"Deepseek-v3 technical report.arXiv preprint arXiv:2412.19437,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.357714Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:f0e716cd89ea0a15875246d782381e8adc71d74c07c74c6098549e8ef254c011","observation_id":"16ee2b9c-a491-44bb-b343-14df6bd663b4","resolution":{"observed_at":"2026-08-06T22:14:19.357714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1806.09055","last_updated":"2019-04-23T06:29:32Z","snapshot_observed_at":"2026-08-01T17:05:53.501765Z","submitted_at":"2018-06-24T00:06:13Z","title":"DARTS: Differentiable Architecture Search","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.09055","snapshot_observed_at":"2026-08-06T22:14:19.416905Z","title":"Darts: Differentiable architecture search","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.416905Z"},"links":{"cited_paper":"/paper/1806.09055","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:136c6fe44b0c8c5c956346017c9d0a43321c505ff54255bb4f218a25807c91f4","observation_id":"b37c74e1-eadf-4001-8422-2a432370ff72","resolution":{"observed_at":"2026-08-06T22:14:19.416905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:22.063718Z","title":"Fineweb-edu: the finest collection of educational content, 2024.https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu","venue":null,"work_id":"5a17e53d-850e-45c7-8bca-1234dcef21fd","year":2024},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.468676Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:67907007da5c56db73e9e067afc811747889f710b49d6d36806db0f17047e38b","observation_id":"45d82e4b-be64-4207-a54c-5c5bbec99369","resolution":{"observed_at":"2026-08-06T22:14:22.111671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.00448","last_updated":"2025-04-14T10:11:13Z","snapshot_observed_at":"2026-07-06T17:10:21.098136Z","submitted_at":"2023-12-31T10:53:58Z","title":"Beyond Chinchilla-Optimal: Accounting for Inference in Language Model Scaling Laws","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.00448","snapshot_observed_at":"2026-08-06T22:14:19.596966Z","title":"Beyond chinchilla-optimal: Accounting for inference in language model scaling laws.arXiv preprint arXiv:2401.00448,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.596966Z"},"links":{"cited_paper":"/paper/2401.00448","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:0e2ed6c75d2463ed8ec699393fc9e97e70c512752180d94ab8a0fcdfbd397f4b","observation_id":"691fbcff-885e-4b63-9d50-80ac488b6cdc","resolution":{"observed_at":"2026-08-06T22:14:19.596966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1611.01232","last_updated":"2017-04-04T19:36:14Z","snapshot_observed_at":"2026-07-06T05:17:20.642708Z","submitted_at":"2016-11-04T00:44:32Z","title":"Deep Information Propagation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.01232","snapshot_observed_at":"2026-08-06T22:14:19.643965Z","title":"Deep information propagation.arXiv preprint arXiv:1611.01232,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.643965Z"},"links":{"cited_paper":"/paper/1611.01232","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:3cfec6150ef07c18d2f6571ae8bb092e6f7a012540c7382dec7b2057f30c2bc1","observation_id":"d1811a32-701b-46bc-9a0f-716da38d2905","resolution":{"observed_at":"2026-08-06T22:14:19.643965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1701.06538","last_updated":"2017-01-23T18:10:00Z","snapshot_observed_at":"2026-07-06T05:27:13.416519Z","submitted_at":"2017-01-23T18:10:00Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1701.06538","snapshot_observed_at":"2026-08-06T22:14:19.690467Z","title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer.arXiv preprint arXiv:1701.06538,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.690467Z"},"links":{"cited_paper":"/paper/1701.06538","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:0f8286316cd4163a50a36fd5822a17a51e63899e38cffa768ac2efac7bece525","observation_id":"b3c29bac-d4fe-4df1-bc35-eb61e82108aa","resolution":{"observed_at":"2026-08-06T22:14:19.690467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.974529Z","title":"Dropout: a simple way to prevent neural networks from overfitting.The journal of machine learning research, 15(1):1929–1958,","venue":null,"work_id":"e5b8330e-5cac-4619-a33a-f83e8ae623ba","year":1929},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.742140Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:4c98907b4c2d3ad3a8772b854ac4a366b85be5147662332f9e6a3b02941483cf","observation_id":"2a4618a6-8bc6-4de1-82ac-58aa2f2a9a07","resolution":{"observed_at":"2026-08-06T22:14:22.010935Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.861929Z","title":"torchtune: Pytorch’s finetuning library, April 2024.https//github.com/ pytorch/torchtune","venue":null,"work_id":"e1290304-d696-4a25-8ae1-27794b703419","year":2024},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.869553Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:eca2ba6f4b50fc56776f3d0d464e3d3d55b135d6cb020669d90fafd8e5563129","observation_id":"66a3de26-abb7-40ef-a9d7-45045dcee1bd","resolution":{"observed_at":"2026-08-06T22:14:21.905199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1611.01578","last_updated":"2017-02-15T05:28:05Z","snapshot_observed_at":"2026-07-06T05:17:29.499249Z","submitted_at":"2016-11-05T00:41:37Z","title":"Neural Architecture Search with Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.01578","snapshot_observed_at":"2026-08-06T22:14:19.952473Z","title":"Neural architecture search with reinforcement learning.arXiv preprint arXiv:1611.01578,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.952473Z"},"links":{"cited_paper":"/paper/1611.01578","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:d5a148afed797b577b73eb2fb3a093d517f7640ea5f1d35912bc6780dea286c4","observation_id":"9c6aa4a5-2d58-45df-9390-c1ea56231cb4","resolution":{"observed_at":"2026-08-06T22:14:19.952473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.747003Z","title":null,"venue":null,"work_id":"14cc8d70-74fe-4e60-a64f-bdb1b450ba0e","year":2023},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:20.040098Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:04e6daefc0da8226030c397656b7902f407ab4c8d88f7b69f1b3e8bd44a34728","observation_id":"126d581e-3eb2-482f-8a24-b3638f6ea7a0","resolution":{"observed_at":"2026-08-06T22:14:21.820871Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.614690Z","title":"B Module Usage and Load Balancing We plot the module usage distribution for all DNA models used in the main text in Fig","venue":null,"work_id":"2e547f09-220d-454d-a87d-e1d7041606ba","year":2016},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:20.045284Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:08a74f26fd57abe08a490ff7c4168151c32a6fa3091827659574a6ed2470f775","observation_id":"ba1c9cb3-de61-4bc8-baaf-a7e2e568a5b4","resolution":{"observed_at":"2026-08-06T22:14:21.678283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.303455Z","title":"The random noise is per-pixel zero-mean, and has a linearly decaying variance, starting at 1 and ending at 0 by the end of the optimization procedure","venue":null,"work_id":"8a237e29-6681-4f2e-b606-7456191e3eab","year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:20.256809Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:e819a2e652c7d0ac74316449e3252314194211ab19fb644424fda6e51569be7a","observation_id":"6395a7bd-f507-47d1-b607-48576b07aec8","resolution":{"observed_at":"2026-08-06T22:14:21.382742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.162301Z","title":"3, we find that the patches following the same path in a randomly initialized model share much greater visual similarities","venue":null,"work_id":"2b9cb57f-a619-488b-9524-324bce5a6ce0","year":2016},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:20.374982Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:66f8e05eb2931b14bd8d5499e9618fdc8a927e1a619bc865ac6aeea57dce5bd3","observation_id":"bbb959d7-42ec-4192-bbe4-8f53e54d3e56","resolution":{"observed_at":"2026-08-06T22:14:21.216489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:21.078321Z","title":"9 is a zoomed-in version of those two figures","venue":null,"work_id":"9784d8ff-5d6e-4c36-b065-10a42fcd2388","year":1966},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:20.461704Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:2d45183d2958119c9dcd92480f873ae074f2a0d010c7329d72bdeb7d28127746","observation_id":"08fb4c44-9031-43fa-a905-51093eb1bc92","resolution":{"observed_at":"2026-08-06T22:14:21.112995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-06T22:14:19.232811Z","title":"Scaling laws for neural language models.arXiv preprint arXiv:2001.08361,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":1991,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.232811Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:fd23cac9f9ac540ad75fa2725c81b09c6565c096b5ddf3e898bd6c7c338e2098","observation_id":"88848d09-fc48-4bfe-9fbd-977fbd2abc37","resolution":{"observed_at":"2026-08-06T22:14:19.232811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:22.171286Z","title":"RACE: Large-scale ReAding comprehension dataset from examinations","venue":null,"work_id":"72ac8f06-dc85-4147-9f88-25cb88b8d117","year":2017},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2012,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.291521Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:7d04692d71310428a3a7731e8a76398ff75f786c72778ecae9e45dd874d7ee82","observation_id":"5161ef2f-64d8-456b-9d59-5bfba7bd16c5","resolution":{"observed_at":"2026-08-06T22:14:22.229680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08524","last_updated":"2025-02-12T16:00:11Z","snapshot_observed_at":"2026-08-07T17:04:51.218878Z","submitted_at":"2025-02-12T16:00:11Z","title":"LLM Pretraining with Continuous Concepts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08524","snapshot_observed_at":"2026-08-06T22:14:19.808871Z","title":"Llm pretraining with continuous concepts.arXiv preprint arXiv:2502.08524,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.808871Z"},"links":{"cited_paper":"/paper/2502.08524","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:d85cfdd1f77e8ff56876fc0a4b33dcdc3dcb712df9c5be5d8a011768575ba675","observation_id":"e9f46559-0637-4498-a104-7320a4abcc83","resolution":{"observed_at":"2026-08-06T22:14:19.808871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-06T22:14:19.192043Z","title":"Distilling the knowledge in a neural network","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.192043Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:535b86a7c514b573999b37c3d3c8a9ca72384da9bcda4e41cbaa5ea105f75dd8","observation_id":"99607442-268c-4eac-9c9c-65012a38d471","resolution":{"observed_at":"2026-08-06T22:14:19.192043Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.01703","last_updated":"2019-12-03T22:06:05Z","snapshot_observed_at":"2026-07-06T08:41:49.632205Z","submitted_at":"2019-12-03T22:06:05Z","title":"PyTorch: An Imperative Style, High-Performance Deep Learning Library","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.01703","snapshot_observed_at":"2026-08-06T22:14:19.516841Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.516841Z"},"links":{"cited_paper":"/paper/1912.01703","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:646bc31b9b55f25140b1eb4d20d6fc0dd29edcc7b666731a0dbc16e4b4152cf4","observation_id":"1d627122-77c7-40cc-9e44-a97146762004","resolution":{"observed_at":"2026-08-06T22:14:19.516841Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:14:18.721573Z","title":"Do language models use their depth efficiently? arXiv preprint arXiv:2505.13898,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.721573Z"},"links":{"citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:b27284b396bcf0b262514fd7db33ef8f452439184377c77fd1a653260443349c","observation_id":"b2f7d93b-bee0-48bd-bfb2-1abc51917bee","resolution":{"observed_at":"2026-08-06T22:14:18.721573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-06T22:14:18.635699Z","title":"Think you have solved question answering? try arc, the ai2 reasoning challenge.arXiv:1803.05457v1,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.635699Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:2b338979ea508f38bfa05ee90844f682a700b8aa975175798b7d952cc018b947","observation_id":"e1b911b4-b597-4b18-8395-38a785d1ba97","resolution":{"observed_at":"2026-08-06T22:14:18.635699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05176","last_updated":"2023-05-09T05:11:02Z","snapshot_observed_at":"2026-07-31T18:33:34.529799Z","submitted_at":"2023-05-09T05:11:02Z","title":"FrugalGPT: How to Use Large Language Models While Reducing Cost and Improving Performance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05176","snapshot_observed_at":"2026-08-06T22:14:18.547802Z","title":"Frugalgpt: How to use large language models while reducing cost and improving performance","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.547802Z"},"links":{"cited_paper":"/paper/2305.05176","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:c9719c5646cd0005fb311f0b969c5665bbcc0275ab2cda15f67f2a5c25babc02","observation_id":"ef8483eb-53bc-4772-9926-b0885c75eb1e","resolution":{"observed_at":"2026-08-06T22:14:18.547802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16710","last_updated":"2024-10-18T04:02:31Z","snapshot_observed_at":"2026-08-08T01:15:44.293937Z","submitted_at":"2024-04-25T16:20:23Z","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16710","snapshot_observed_at":"2026-08-06T22:14:18.881708Z","title":"Layerskip: Enabling early exit inference and self-speculative decoding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.881708Z"},"links":{"cited_paper":"/paper/2404.16710","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:e232195438592a630c85f042c6f865f26ffe994e529ff39832d868aef6333c05","observation_id":"9122c42c-cc6e-464e-84c7-8063b72564ae","resolution":{"observed_at":"2026-08-06T22:14:18.881708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T22:14:19.055624Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.055624Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:d95729381a3e039867f298bba0178c2f07a85cdd2d19ff7060523d1f2ebdc859","observation_id":"538acbe2-ac9c-413a-928f-44ade83f4014","resolution":{"observed_at":"2026-08-06T22:14:19.055624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1308.3432","last_updated":"2013-08-15T15:19:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2013-08-15T15:19:34Z","title":"Estimating or Propagating Gradients Through Stochastic Neurons for Conditional Computation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1308.3432","snapshot_observed_at":"2026-08-06T22:14:18.461421Z","title":"Estimating or propagating gradients through stochastic neurons for conditional computation.arXiv preprint arXiv:1308.3432,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:18.461421Z"},"links":{"cited_paper":"/paper/1308.3432","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:2550ef36d1038277b07158a83a524b88f323093b02d5d34bafbeb27b626f25cf","observation_id":"1e1d56b4-b56f-412e-8ea8-f24089f2cff0","resolution":{"observed_at":"2026-08-06T22:14:18.461421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T22:14:19.114883Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.114883Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:18250ab64bb5b07aae3d5101af17181169c1b83ed02b077271b04be9ed7f087a","observation_id":"a0a084e5-e6ef-4663-85d2-3565f95c0be8","resolution":{"observed_at":"2026-08-06T22:14:19.114883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06727","last_updated":"2022-12-13T16:55:12Z","snapshot_observed_at":"2026-07-06T14:30:08.686540Z","submitted_at":"2022-12-13T16:55:12Z","title":"What do Vision Transformers Learn? A Visual Exploration","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06727","snapshot_observed_at":"2026-08-06T22:14:19.008165Z","title":"What do vision transformers learn? a visual exploration.arXiv preprint arXiv:2212.06727,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.008165Z"},"links":{"cited_paper":"/paper/2212.06727","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:d1ea144ece8f9018abc201c4ae13e3b3d6df90b09654f6fe49a3f4aeb2de6eff","observation_id":"98349d3c-4cf6-4b1e-bfa9-44ccd5c843da","resolution":{"observed_at":"2026-08-06T22:14:19.008165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T00:16:10.504751Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":0,"verified_fuzzy":9},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 0 inbound Pith citation observations for arXiv:2506.22389."}