{"as_of":"2026-08-11T11:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e79e0660f01b3b6f7858135a2adc6ffa990cf8e2dfeb968914fd26812a8da5d9","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T04:15:20.027659Z","state":"measured"},{"denominator":139,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":139,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":195,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T22:50:28.835675Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":12,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2308.01390","last_updated":"2023-08-07T17:53:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-02T19:10:23Z","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-14T01:52:01.163900Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2308.01390"},"observation_digest":"sha256:3658cf982d9436c64cafcc03d180706448f2ef9af2ceb28777f9662cbec7ddd7","observation_id":"b5d7c391-6a4e-4b27-9d4f-5e4f789ece4f","resolution":{"observed_at":"2026-05-14T01:52:01.372311Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2309.00071","last_updated":"2026-02-06T19:40:50Z","snapshot_observed_at":"2026-08-01T02:15:47.181936Z","submitted_at":"2023-08-31T18:18:07Z","title":"YaRN: Efficient Context Window Extension of Large Language Models","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T06:46:50.042423Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2309.00071"},"observation_digest":"sha256:369852f32a93e61389c85a16b6428e974ff5e73d75f4ddb056a6545576017f97","observation_id":"da2b49d5-1dd8-4c5f-ac93-510092fe9813","resolution":{"observed_at":"2026-05-12T06:46:51.302758Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2402.09353","last_updated":"2024-07-09T05:59:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-14T17:59:34Z","title":"DoRA: Weight-Decomposed Low-Rank Adaptation","version":6},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-05-15T22:26:21.449135Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2402.09353"},"observation_digest":"sha256:970956a2626e4949c6e75f329fe0dad5eddace5fad5e818f0be3df59450344d4","observation_id":"e0036ecd-cc11-4c26-9552-dc9b0b9316a7","resolution":{"observed_at":"2026-05-15T22:26:21.578949Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2403.07691","last_updated":"2024-03-14T07:47:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T14:34:08Z","title":"ORPO: Monolithic Preference Optimization without Reference Model","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-16T09:34:04.394588Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2403.07691"},"observation_digest":"sha256:5625728815ecd0995e4165e49494030d86cd015ab6be87b5960d1da2c9f75cc7","observation_id":"68b7e74b-f73b-48cc-b83b-c40a0854b682","resolution":{"observed_at":"2026-05-16T09:34:04.722924Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-15T03:28:57.008431Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2405.11143"},"observation_digest":"sha256:786dbe130b805e25f05b991788e2b917f0135bfb26e34ee7fdc7f07d313f15aa","observation_id":"ac7acae5-897e-4548-b60e-6160e5c2a5f5","resolution":{"observed_at":"2026-05-15T03:28:57.198747Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2406.04093","last_updated":"2024-06-06T14:10:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-06T14:10:12Z","title":"Scaling and evaluating sparse autoencoders","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-12T17:47:23.089288Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2406.04093"},"observation_digest":"sha256:9f651aaa469afbbd29dfd8bbced4b64820b3920bf2034d1e54edef90943a9563","observation_id":"4556994b-460e-4176-87fb-cc83996da08a","resolution":{"observed_at":"2026-05-12T17:47:23.336609Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2406.06525","last_updated":"2024-06-10T17:59:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-10T17:59:52Z","title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-11T22:09:16.622717Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2406.06525"},"observation_digest":"sha256:7115801ce3ce1de1f29ba50d5804776030c2ff271b0bc997e5d731f7d075ebd4","observation_id":"b2b399e8-7da7-4e3a-87cd-eccd0d2e7cbe","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2406.08334","last_updated":"2026-04-20T05:53:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-12T15:40:06Z","title":"ProTrain: Efficient LLM Training via Memory-Aware Techniques","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-23T23:48:40.669335Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2406.08334"},"observation_digest":"sha256:ab5d87855a4b96a195502736bde5e3063da009959d52b1476be67ed7907996d8","observation_id":"a85e440b-ea3f-4fca-b118-431cb8691d8e","resolution":{"observed_at":"2026-05-23T23:53:39.373647Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2406.09246","last_updated":"2024-09-05T19:46:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-13T15:46:55Z","title":"OpenVLA: An Open-Source Vision-Language-Action Model","version":3},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-10T14:46:35.942338Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2406.09246"},"observation_digest":"sha256:adbd93be91eb230ef0e771a42145972550226d7f054d6fd4f135ad9669c5ae24","observation_id":"0548510b-2da9-4a7f-a398-8612ed8eed49","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"reference_index":150,"source":"pdf_text","source_observed_at":"2026-05-17T00:05:03.547664Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2406.16860"},"observation_digest":"sha256:ad9263e348ace82f30bddc504eb7f6191267e50b1aed89a6214250dc87ba3a5f","observation_id":"2ab3bbfd-16b3-4d44-bbac-79476d2ab9ea","resolution":{"observed_at":"2026-05-17T00:05:03.871857Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"reference_index":135,"source":"pdf_text","source_observed_at":"2026-05-15T01:55:12.501409Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2409.17146"},"observation_digest":"sha256:a7d4b211d22d24fc39f7a030478ce43fba4139c7573abb3681e6c19778ebc17d","observation_id":"c1d1b286-c181-4047-9ad0-95940478d3ed","resolution":{"observed_at":"2026-05-15T01:55:12.715416Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2410.13720","last_updated":"2025-02-26T16:05:55Z","snapshot_observed_at":"2026-08-10T21:41:31.247717Z","submitted_at":"2024-10-17T16:22:46Z","title":"Movie Gen: A Cast of Media Foundation Models","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-11T14:16:18.521699Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2410.13720"},"observation_digest":"sha256:ab31a1a9c7789d71ca9a59512d7853796d05d8b238838cdcaa608015ba9f7f46","observation_id":"8e6105ec-4847-4d40-814d-98941fcfc30b","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2410.15155","last_updated":"2026-04-26T22:56:07Z","snapshot_observed_at":"2026-07-06T19:36:27.984192Z","submitted_at":"2024-10-19T16:58:34Z","title":"On the Convergence Theory of Pipeline Gradient-based Analog In-memory Training","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-23T19:15:54.807005Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2410.15155"},"observation_digest":"sha256:13338537d458cea47d2a7f4fd3bcc803ca249a83018beb0f2a14ee875fc5b8b9","observation_id":"572ed9ce-d888-4211-a562-39aa1b12c217","resolution":{"observed_at":"2026-05-23T19:18:20.805241Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2412.00131","last_updated":"2024-11-28T14:07:45Z","snapshot_observed_at":"2026-08-11T04:49:22.222888Z","submitted_at":"2024-11-28T14:07:45Z","title":"Open-Sora Plan: Open-Source Large Video Generation Model","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T08:38:27.946746Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2412.00131"},"observation_digest":"sha256:bfc4c37aac1bacb6be6473f57f8a552cd1de652b403f232393ccd4283847be7a","observation_id":"517148f7-01cb-45a4-8633-55d02be40235","resolution":{"observed_at":"2026-05-23T08:42:45.224881Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2412.14164","last_updated":"2024-12-18T18:58:50Z","snapshot_observed_at":"2026-08-08T15:05:21.947334Z","submitted_at":"2024-12-18T18:58:50Z","title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","version":1},"reference_index":196,"source":"arxiv_source","source_observed_at":"2026-05-17T07:51:12.953777Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2412.14164"},"observation_digest":"sha256:81ae39497c296915dabdbfb3b15e221d5a896bf4d995401b372f4673256edbb5","observation_id":"a0103f0b-3489-4258-bc5c-4ca155d23123","resolution":{"observed_at":"2026-05-17T07:51:13.125695Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2412.15689","last_updated":"2026-05-06T21:36:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-20T09:07:36Z","title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-23T06:57:50.897865Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2412.15689"},"observation_digest":"sha256:6b367f828c10a60334d9bcd4bc226ea104e6696c305f856934d90e56a06d4aa8","observation_id":"4f123731-f06b-4490-ad4e-0be9ed4e9d74","resolution":{"observed_at":"2026-05-23T07:02:41.794094Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T22:50:28.835675Z","title":"tX i=1 dlt dyt λt,i(Ai θhi−1 + Bi θ ˆxi) # + dlt dyt Ct θht =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.00692","last_updated":"2025-01-01T01:10:59Z","snapshot_observed_at":"2026-08-10T22:42:25.533894Z","submitted_at":"2025-01-01T01:10:59Z","title":"Adjoint sharding for very long context training of state space models","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T22:50:28.835675Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.00692"},"observation_digest":"sha256:b2507ff7ea5d14b6e05ebf49bd555f1d3ddd41a754abac6553b5503982f2f16a","observation_id":"b940b398-90de-4f1e-8780-c9364d5ff600","resolution":{"observed_at":"2026-08-10T22:50:28.835675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T22:45:11.209703Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.00874","last_updated":"2025-05-07T18:16:42Z","snapshot_observed_at":"2026-08-11T02:41:48.620951Z","submitted_at":"2025-01-01T15:43:07Z","title":"LUSIFER: Language Universal Space Integration for Enhanced Multilingual Embeddings with Large Language Models","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T22:45:11.209703Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.00874"},"observation_digest":"sha256:fbe3c0c5b91bcac960d21a27fb8b945d69241229f3e5be42c861878a67216d9a","observation_id":"83b24d57-cbc2-43b2-9953-7ed3c9ce7f7a","resolution":{"observed_at":"2026-08-10T22:45:11.209703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T22:20:59.490966Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.01951","last_updated":"2025-02-24T22:02:47Z","snapshot_observed_at":"2026-08-10T22:12:19.854787Z","submitted_at":"2025-01-03T18:54:46Z","title":"MixGCN: Scalable GCN Training by Mixture of Parallelism and Mixture of Accelerators","version":3},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-10T22:20:59.490966Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.01951"},"observation_digest":"sha256:df495ad3ee3d87f1081eccd74e7b2e7d4546abad0fd69b1399daf95424d70c0a","observation_id":"354e0a3f-1189-409a-8e77-fc09bf6937e8","resolution":{"observed_at":"2026-08-10T22:20:59.490966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T21:42:38.210400Z","title":"Available: https://arxiv.org/abs/2304.11277","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.04266","last_updated":"2025-02-04T04:32:42Z","snapshot_observed_at":"2026-08-10T21:35:13.113483Z","submitted_at":"2025-01-08T04:19:57Z","title":"Scaling Large Language Model Training on Frontier with Low-Bandwidth Partitioning","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T21:42:38.210400Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.04266"},"observation_digest":"sha256:845a816ee90e80824971a6d182b237dceb3a51d4a56887a3af6579ac426a3741","observation_id":"2b629f5e-1470-412e-a753-13dd96880d73","resolution":{"observed_at":"2026-08-10T21:42:38.210400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T21:17:58.463543Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.05450","last_updated":"2025-01-10T18:58:11Z","snapshot_observed_at":"2026-08-11T06:05:04.263386Z","submitted_at":"2025-01-09T18:59:56Z","title":"Decentralized Diffusion Models","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T21:17:58.463543Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.05450"},"observation_digest":"sha256:43b207160221be3b5306fe94dbcf6971298008f2794047f7845407788d1f2aa1","observation_id":"c6644dc4-0911-45b3-af0e-0d4be94538c3","resolution":{"observed_at":"2026-08-10T21:17:58.463543Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T20:53:56.289464Z","title":"arXiv preprint arXiv:2304.11277","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.06741","last_updated":"2025-01-12T07:30:49Z","snapshot_observed_at":"2026-08-10T20:48:17.305245Z","submitted_at":"2025-01-12T07:30:49Z","title":"Hierarchical Divide-and-Conquer for Fine-Grained Alignment in LLM-Based Medical Evaluation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T20:53:56.289464Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.06741"},"observation_digest":"sha256:97109058578a15fe4088a78ce0efdb41fa1e5473cb72bf344f16efa7dd9f5ac3","observation_id":"49dad49a-2057-42ca-a7c3-01ac5d9b86ab","resolution":{"observed_at":"2026-08-10T20:53:56.289464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T20:41:57.870308Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.07721","last_updated":"2025-01-13T22:14:45Z","snapshot_observed_at":"2026-08-10T20:34:30.401279Z","submitted_at":"2025-01-13T22:14:45Z","title":"LLMic: Romanian Foundation Language Model","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T20:41:57.870308Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.07721"},"observation_digest":"sha256:fa9e7dd35e08b58ee1e918564b94f84d13e2fa6bd37ad9f9d84a049e78089ca1","observation_id":"45b7dec0-19ee-429c-9aa7-e8bf2fd249ab","resolution":{"observed_at":"2026-08-10T20:41:57.870308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T17:58:50.573400Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11771","last_updated":"2025-08-13T19:33:58Z","snapshot_observed_at":"2026-08-11T08:13:19.499819Z","submitted_at":"2025-01-20T22:23:50Z","title":"Characterization of GPU TEE Overheads in Distributed Data Parallel ML Training","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T17:58:50.573400Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.11771"},"observation_digest":"sha256:336dcdcab8a5729409e05e454a4736a634ddb7d0259dc6e8fe473b873b1d3f94","observation_id":"d47e6977-df76-49ce-a46d-39e24005b275","resolution":{"observed_at":"2026-08-10T17:58:50.573400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T18:01:46.429279Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.11779","last_updated":"2025-02-11T17:36:32Z","snapshot_observed_at":"2026-08-11T09:05:01.680844Z","submitted_at":"2025-01-20T23:10:13Z","title":"Glinthawk: A Two-Tiered Architecture for Offline LLM Inference","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T18:01:46.429279Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.11779"},"observation_digest":"sha256:a3294ff74afabfa3fde91dfcaccaff1d811cb962b87af5397f3e87ac25132357","observation_id":"ead83c85-b29d-4a54-8b74-e380f9d34488","resolution":{"observed_at":"2026-08-10T18:01:46.429279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-10T21:37:14.570052Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13111","last_updated":"2025-01-08T14:38:13Z","snapshot_observed_at":"2026-08-10T21:27:53.710306Z","submitted_at":"2025-01-08T14:38:13Z","title":"iServe: An Intent-based Serving System for LLMs","version":1},"reference_index":129,"source":"pdf_text","source_observed_at":"2026-08-10T21:37:14.570052Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.13111"},"observation_digest":"sha256:15daa7e58886c7a719a7724b2c49a6a615889fbedd48b7d68a79bb215c4472c3","observation_id":"7f83f168-eb42-40e9-99b0-7b3cde396051","resolution":{"observed_at":"2026-08-10T21:37:14.570052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-09T23:17:19.897463Z","title":"27 Streaming DiLoCo with overlapping communication: Towards a Distributed Free Lunch Supplementary Materials Architecture hyperparameters","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.18512","last_updated":"2025-01-30T17:23:50Z","snapshot_observed_at":"2026-08-10T13:40:15.088522Z","submitted_at":"2025-01-30T17:23:50Z","title":"Streaming DiLoCo with overlapping communication: Towards a Distributed Free Lunch","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-09T23:17:19.897463Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2501.18512"},"observation_digest":"sha256:357f1f47610447dbcb8b198cf2f943b6e4d6d9765cf1ca46d688ae23e0385560","observation_id":"2937bea6-2252-4bee-9bc3-21c5a240be25","resolution":{"observed_at":"2026-08-09T23:17:19.897463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-08T22:25:11.259488Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04562","last_updated":"2025-02-06T23:29:32Z","snapshot_observed_at":"2026-08-09T15:16:36.174055Z","submitted_at":"2025-02-06T23:29:32Z","title":"Mixture of neural operator experts for learning boundary conditions and model selection","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-08T22:25:11.259488Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2502.04562"},"observation_digest":"sha256:64e8aa066dd565e8e42a3abd40b692e54df7fa1ffe422a0138a928fe7bc3c8ba","observation_id":"d624f376-44a0-4e63-a64b-555cc136f2e6","resolution":{"observed_at":"2026-08-08T22:25:11.259488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-08T14:50:58.981760Z","title":"Pytorch fsdp: Experiences on scaling fully sharded data parallel, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06635","last_updated":"2025-02-13T07:31:55Z","snapshot_observed_at":"2026-08-09T03:32:09.083141Z","submitted_at":"2025-02-10T16:31:37Z","title":"Steel-LLM:From Scratch to Open Source -- A Personal Journey in Building a Chinese-Centric LLM","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-08T14:50:58.981760Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2502.06635"},"observation_digest":"sha256:3d64627f9ab313e1fe30b4732673262ab9a5d547a0537ff71c110c5ce50384d4","observation_id":"afc1e85c-e3b7-454d-a9e5-ed011e8d30ce","resolution":{"observed_at":"2026-08-08T14:50:58.981760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-08T14:52:07.603608Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06643","last_updated":"2025-02-10T16:34:36Z","snapshot_observed_at":"2026-08-09T20:43:08.168062Z","submitted_at":"2025-02-10T16:34:36Z","title":"MoETuner: Optimized Mixture of Expert Serving with Balanced Expert Placement and Token Routing","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-08T14:52:07.603608Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2502.06643"},"observation_digest":"sha256:1e12c1e931c1556531b0efcbed8b462f7dbfc67cddc820025edfdc0dfc59cf5c","observation_id":"d086f6f5-ea3f-4c28-9a85-b5e8e452f58d","resolution":{"observed_at":"2026-08-08T14:52:07.603608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-08T14:26:46.887727Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06782","last_updated":"2025-02-12T10:07:07Z","snapshot_observed_at":"2026-08-08T14:18:27.556352Z","submitted_at":"2025-02-10T18:58:11Z","title":"Lumina-Video: Efficient and Flexible Video Generation with Multi-scale Next-DiT","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-08T14:26:46.887727Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2502.06782"},"observation_digest":"sha256:4a29e65029aeb899558797bbb742197e3cb9ce98bf9a318a68eae25ff4d682a6","observation_id":"0d6ba78c-503e-4bca-9d2a-5947c2a03589","resolution":{"observed_at":"2026-08-08T14:26:46.887727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-08T12:25:42.771337Z","title":"Pytorch FSDP: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.07563","last_updated":"2025-02-11T14:01:39Z","snapshot_observed_at":"2026-08-08T12:15:35.512720Z","submitted_at":"2025-02-11T14:01:39Z","title":"LASP-2: Rethinking Sequence Parallelism for Linear Attention and Its Hybrid","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T12:25:42.771337Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2502.07563"},"observation_digest":"sha256:3279a462d3462d5386d8694463b3ef35ea8d6116e25f0e120cc7edd691f314cd","observation_id":"63a8ebcd-779d-4711-b25d-16f3495823ca","resolution":{"observed_at":"2026-08-08T12:25:42.771337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2503.07703","last_updated":"2025-03-10T17:58:33Z","snapshot_observed_at":"2026-07-06T20:50:09.622181Z","submitted_at":"2025-03-10T17:58:33Z","title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T08:27:36.242416Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2503.07703"},"observation_digest":"sha256:f624c6e45912cc2e7d0ea2598b8c731faa24ba626d537a315727fb1acf8135b0","observation_id":"cff3e95f-3b22-4765-bdd7-5b9ae15d0571","resolution":{"observed_at":"2026-05-17T08:27:36.278797Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2505.13211","last_updated":"2025-05-19T14:58:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-19T14:58:50Z","title":"MAGI-1: Autoregressive Video Generation at Scale","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T20:31:15.700943Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2505.13211"},"observation_digest":"sha256:42cf9e144d238df4c44fdde992e02f32562988d5594fd30b3eb2a1024325ab82","observation_id":"c4530406-ad43-4014-96c3-6cb0079e9acc","resolution":{"observed_at":"2026-05-13T20:31:15.750971Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T15:06:08.308487Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16326","last_updated":"2025-08-04T06:14:13Z","snapshot_observed_at":"2026-08-10T07:26:23.903392Z","submitted_at":"2025-05-22T07:32:17Z","title":"ChemMLLM: Chemical Multimodal Large Language Model","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T15:06:08.308487Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2505.16326"},"observation_digest":"sha256:2b87a8cccce12a329a6f4d1a8c3f0f71dccc2a68a8c90d0285dc6f816bee6606","observation_id":"4bd090f3-fcb5-40dd-b93b-034c8917c9d8","resolution":{"observed_at":"2026-08-07T15:06:08.308487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T15:09:16.870871Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16363","last_updated":"2025-05-22T08:16:48Z","snapshot_observed_at":"2026-08-07T15:00:34.424850Z","submitted_at":"2025-05-22T08:16:48Z","title":"AdamS: Momentum Itself Can Be A Normalizer for LLM Pretraining and Post-training","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T15:09:16.870871Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2505.16363"},"observation_digest":"sha256:fc2194c9a598c514e02255132058b0996b4d578f2e0ed93fe4a326177cb0eef4","observation_id":"e783c794-8549-43fc-91f5-ba780525bd61","resolution":{"observed_at":"2026-08-07T15:09:16.870871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2505.18719","last_updated":"2025-05-24T14:42:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-24T14:42:51Z","title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-05-16T12:55:40.245908Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2505.18719"},"observation_digest":"sha256:1a5c185d712ece02baa170dec45a03346646aaa0d5e65c5cfcc2d3e54e4797f3","observation_id":"65a0e00b-9d5f-44d8-9679-92277947ce43","resolution":{"observed_at":"2026-05-16T12:55:40.402013Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T14:17:13.922316Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19529","last_updated":"2025-05-29T16:57:36Z","snapshot_observed_at":"2026-08-10T01:35:01.414848Z","submitted_at":"2025-05-26T05:29:47Z","title":"Small Language Models: Architectures, Techniques, Evaluation, Problems and Future Adaptation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:17:13.922316Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2505.19529"},"observation_digest":"sha256:e7e7e0e0f8adb07c3e2dc240ac2e34f11fc6b49c18253d78f4da8400cdbf876e","observation_id":"bd81e855-636f-4768-bbaf-40464cf7f24e","resolution":{"observed_at":"2026-08-07T14:17:13.922316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T13:10:11.740399Z","title":"PyTorch FSDP: Experi- ences on Scaling Fully Sharded Data Parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22664","last_updated":"2025-08-02T22:03:19Z","snapshot_observed_at":"2026-08-08T08:57:24.731047Z","submitted_at":"2025-05-28T17:59:59Z","title":"Zero-Shot Vision Encoder Grafting via LLM Surrogates","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T13:10:11.740399Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2505.22664"},"observation_digest":"sha256:ab2e6a06293ca50487c913a816979c57749654f018f8f2b761119ff09396b444","observation_id":"f1dca64a-039e-41c8-b879-848cd520fce3","resolution":{"observed_at":"2026-08-07T13:10:11.740399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T12:58:05.511142Z","title":"Available: https://arxiv .org/abs/2304.11277","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.23072","last_updated":"2025-05-29T04:24:56Z","snapshot_observed_at":"2026-08-09T20:15:37.848968Z","submitted_at":"2025-05-29T04:24:56Z","title":"Speeding up Model Loading with fastsafetensors","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T12:58:05.511142Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2505.23072"},"observation_digest":"sha256:dd9bc4efa3ce3ca7cfdd07030f625124886209f33befa1364454f16a8d26762c","observation_id":"5d4bbd67-4c41-4ccf-be99-f739761dd62b","resolution":{"observed_at":"2026-08-07T12:58:05.511142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T11:55:15.412077Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01260","last_updated":"2026-07-09T08:54:09Z","snapshot_observed_at":"2026-08-10T06:04:27.598058Z","submitted_at":"2025-06-02T02:19:22Z","title":"Subspace Networks: Scaling Decentralized Training with Communication-Efficient Model Parallelism","version":3},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-07T11:55:15.412077Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.01260"},"observation_digest":"sha256:b647aa47260175ddc3222d48054202a722461f2c1e46c8be77894c71f784d978","observation_id":"03303144-9887-4265-b371-af3dcdde92c0","resolution":{"observed_at":"2026-08-07T11:55:15.412077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T11:40:23.793987Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-09T11:24:14.753215Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T11:40:23.793987Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:4a14b16ed7bebe8f8d544b3db5f7923761d374be8ed84b73a08c04924728b2eb","observation_id":"6b8bbcb3-4434-4ac0-a050-371f282b13aa","resolution":{"observed_at":"2026-08-07T11:40:23.793987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T11:33:57.305024Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel.arXiv preprint arXiv:2304.11277,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.02177","last_updated":"2025-06-02T19:03:00Z","snapshot_observed_at":"2026-08-09T22:08:47.391957Z","submitted_at":"2025-06-02T19:03:00Z","title":"Act Only When It Pays: Efficient Reinforcement Learning for LLM Reasoning via Selective Rollouts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:33:57.305024Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.02177"},"observation_digest":"sha256:044fab58066bae451a4807e9b978b3111b5922c5b51bab572296513ce0699b67","observation_id":"af96af09-a3de-4845-a58d-785ada5276b4","resolution":{"observed_at":"2026-08-07T11:33:57.305024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T11:20:51.209813Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02787","last_updated":"2025-06-03T12:14:17Z","snapshot_observed_at":"2026-08-10T02:22:46.475901Z","submitted_at":"2025-06-03T12:14:17Z","title":"Rethinking Dynamic Networks and Heterogeneous Computing with Automatic Parallelization","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:20:51.209813Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.02787"},"observation_digest":"sha256:6c1a9740c7ed80c003dad8cb9f0cc673a36293475c1a507624f32d370770e4f3","observation_id":"16d7f9f4-8772-4e38-ae88-1376079cd462","resolution":{"observed_at":"2026-08-07T11:20:51.209813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T10:28:35.727661Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05343","last_updated":"2025-06-11T15:48:38Z","snapshot_observed_at":"2026-08-09T14:23:10.699088Z","submitted_at":"2025-06-05T17:59:54Z","title":"ContentV: Efficient Training of Video Generation Models with Limited Compute","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:35.727661Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.05343"},"observation_digest":"sha256:5c0d2996e409e090b33f9f44cb7f3ee0318cbb7a0ba44abe8653fe86b220a2ec","observation_id":"08cf2c8d-73bb-46a5-aa27-40810d94b887","resolution":{"observed_at":"2026-08-07T10:28:35.727661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T06:02:33.064390Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06122","last_updated":"2025-06-06T14:33:56Z","snapshot_observed_at":"2026-08-07T21:55:43.371585Z","submitted_at":"2025-06-06T14:33:56Z","title":"Reinforcement Learning Optimization for Large-Scale Learning: An Efficient and User-Friendly Scaling Library","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-07T06:02:33.064390Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.06122"},"observation_digest":"sha256:fdc995d9e06044e485687e180cf812f7f4e2f8b1a0afed577ae6c7ded0cfaae0","observation_id":"6dc93f18-6194-4cba-bd60-729ece3974ec","resolution":{"observed_at":"2026-08-07T06:02:33.064390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T05:25:18.801426Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08132","last_updated":"2025-06-09T18:35:18Z","snapshot_observed_at":"2026-08-09T05:47:31.693546Z","submitted_at":"2025-06-09T18:35:18Z","title":"Congestion-Aware Path Selection for Load Balancing in AI Clusters","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T05:25:18.801426Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.08132"},"observation_digest":"sha256:72436d9e8b5c02edec04460c4c3ec18b166572f77c0dba871588cc06a600b1b8","observation_id":"e0e74a71-5549-4498-b90b-3a0dd6444d4f","resolution":{"observed_at":"2026-08-07T05:25:18.801426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-07T05:20:51.235247Z","title":"Pytorch fsdp: experi- ences on scaling fully sharded data parallel.arXiv preprint arXiv:2304.11277, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08210","last_updated":"2025-06-09T20:29:53Z","snapshot_observed_at":"2026-08-10T12:36:56.309400Z","submitted_at":"2025-06-09T20:29:53Z","title":"A Comprehensive Study of Decoder-Only LLMs for Text-to-Image Generation","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T05:20:51.235247Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.08210"},"observation_digest":"sha256:23c0fd80f8ad96615c248ee6d819c10bd2a06eecf6594cbd93a2c9b267986b20","observation_id":"d7afd82b-89f1-4642-9c27-61a1bdf32373","resolution":{"observed_at":"2026-08-07T05:20:51.235247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2506.09113","last_updated":"2025-06-28T11:58:23Z","snapshot_observed_at":"2026-08-09T02:36:54.979612Z","submitted_at":"2025-06-10T17:56:11Z","title":"Seedance 1.0: Exploring the Boundaries of Video Generation Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-11T12:09:56.836351Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.09113"},"observation_digest":"sha256:731dd468feb9fed81c5036aae42022a616ba5c5d385874b9fc5d55667622cd66","observation_id":"9693bc11-1736-491c-8052-2ac7ee014da0","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T22:29:59.851666Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21411","last_updated":"2025-06-26T15:58:14Z","snapshot_observed_at":"2026-08-09T22:15:23.287627Z","submitted_at":"2025-06-26T15:58:14Z","title":"Distributed Cross-Channel Hierarchical Aggregation for Foundation Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:29:59.851666Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.21411"},"observation_digest":"sha256:9de1f83ed0b07b0b22cfadb1d77e668ce6355ced74caaf49fa26d3799f6451e2","observation_id":"3de90bba-28ee-4bcb-af29-0399a6d7efb1","resolution":{"observed_at":"2026-08-06T22:29:59.851666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T22:31:39.220992Z","title":"PyTorch FSDP: Experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21468","last_updated":"2025-06-26T16:56:43Z","snapshot_observed_at":"2026-08-09T17:02:44.116383Z","submitted_at":"2025-06-26T16:56:43Z","title":"TopK Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T22:31:39.220992Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2506.21468"},"observation_digest":"sha256:adbe63affa5cd918379fdff6d42e07d0f4e215d90646e31aa4befb0ec1b3918e","observation_id":"757b2521-884d-4004-a072-b40518b8c712","resolution":{"observed_at":"2026-08-06T22:31:39.220992Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T20:23:50.397044Z","title":"Available: https://arxiv .org/abs/2304.11277","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.03114","last_updated":"2025-07-03T18:53:03Z","snapshot_observed_at":"2026-08-10T15:46:36.098464Z","submitted_at":"2025-07-03T18:53:03Z","title":"Characterizing Compute-Communication Overlap in GPU-Accelerated Distributed Deep Learning: Performance and Power Implications","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T20:23:50.397044Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.03114"},"observation_digest":"sha256:3ea173a670944cdf0ca77f4b031f34ed83b21805bacebe20c976b10eedf5cbc4","observation_id":"79ec7187-4f1e-43aa-989f-c5c797e7a15e","resolution":{"observed_at":"2026-08-06T20:23:50.397044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T18:33:50.454819Z","title":"arXiv preprint arXiv:2304.11277 (2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08119","last_updated":"2025-07-10T19:14:16Z","snapshot_observed_at":"2026-08-08T15:14:32.634031Z","submitted_at":"2025-07-10T19:14:16Z","title":"Photonic Rails in ML Datacenters","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T18:33:50.454819Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.08119"},"observation_digest":"sha256:9b09d441e0bf76a95ba2fb266414db5e7d4ace65aa7055473cab47a77d83caca","observation_id":"5ee06caa-f5a1-45f7-a2e0-9373b41413d8","resolution":{"observed_at":"2026-08-06T18:33:50.454819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2507.09025","last_updated":"2026-04-18T10:36:33Z","snapshot_observed_at":"2026-08-11T05:45:50.806515Z","submitted_at":"2025-07-11T21:19:18Z","title":"Lizard: An Efficient Linearization Framework for Large Language Models","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-19T04:37:55.034479Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.09025"},"observation_digest":"sha256:83227dfb44f57cd06a46f47ef904ef895c27da608768416e3dd3555951f76475","observation_id":"8e3887bc-3eb4-4155-85eb-603d11156688","resolution":{"observed_at":"2026-05-19T04:42:04.548123Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T18:19:59.473388Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09029","last_updated":"2026-05-31T02:40:58Z","snapshot_observed_at":"2026-08-09T07:04:00.337129Z","submitted_at":"2025-07-11T21:25:11Z","title":"Model Parallelism With Subnetwork Data Parallelism","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T18:19:59.473388Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.09029"},"observation_digest":"sha256:94d7e38fbb3de22bac3a673893b198b52d62aab7f28bfc233fd184b91fd82a28","observation_id":"b9c5ddec-5c58-4f7b-aaec-d6143cd8e4f1","resolution":{"observed_at":"2026-08-06T18:19:59.473388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T17:52:50.798959Z","title":"PyTorch FSDP : Experiences on Scaling Fully Sharded Data Parallel , September 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10618","last_updated":"2025-07-13T21:28:02Z","snapshot_observed_at":"2026-08-08T03:18:40.989863Z","submitted_at":"2025-07-13T21:28:02Z","title":"Compute Requirements for Algorithmic Innovation in Frontier AI Models","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T17:52:50.798959Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.10618"},"observation_digest":"sha256:da49744d85cc5aa868a041ffc01c67a5b2f4a68ae60eeeb4b537e6909821e646","observation_id":"5dc1ac43-38ad-440f-8dcf-e4a9f8245472","resolution":{"observed_at":"2026-08-06T17:52:50.798959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T19:08:27.545460Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.11544","last_updated":"2025-07-08T23:58:01Z","snapshot_observed_at":"2026-08-06T22:09:08.446996Z","submitted_at":"2025-07-08T23:58:01Z","title":"The Safety Gap Toolkit: Evaluating Hidden Dangers of Open-Source Models","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T19:08:27.545460Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.11544"},"observation_digest":"sha256:c9026f824880dd9282651a8a36c544fefe66c466b47b93022797d3dc38310325","observation_id":"c04afad8-82fe-4773-8d67-97a4c6c89e56","resolution":{"observed_at":"2026-08-06T19:08:27.545460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T14:43:27.268753Z","title":"Pytorch fsdp: Experiences on scaling fully sharded data parallel, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18013","last_updated":"2025-07-29T10:30:18Z","snapshot_observed_at":"2026-08-07T23:55:13.139896Z","submitted_at":"2025-07-24T01:00:48Z","title":"Technical Report of TeleChat2, TeleChat2.5 and T1","version":3},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T14:43:27.268753Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.18013"},"observation_digest":"sha256:a14f4c91f8a4d1718b0857decdee3595039eae090d6096f0663fb7cfdaba3271","observation_id":"6bb105cb-675b-4ebd-9ad6-694882815760","resolution":{"observed_at":"2026-08-06T14:43:27.268753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T14:01:41.855087Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19845","last_updated":"2025-07-26T07:39:30Z","snapshot_observed_at":"2026-08-09T15:03:43.580164Z","submitted_at":"2025-07-26T07:39:30Z","title":"MegatronApp: Efficient and Comprehensive Management on Distributed LLM Training","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T14:01:41.855087Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.19845"},"observation_digest":"sha256:bf0fb09bbe8aa9c8a67cc89c29f7f229796a34004a25fcc2dd10d646fa3ab66b","observation_id":"93504845-a558-4545-a527-62a941a443c7","resolution":{"observed_at":"2026-08-06T14:01:41.855087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-06T15:02:03.011140Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21138","last_updated":"2025-07-22T23:57:11Z","snapshot_observed_at":"2026-08-09T08:43:35.449807Z","submitted_at":"2025-07-22T23:57:11Z","title":"TTS-1 Technical Report","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T15:02:03.011140Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2507.21138"},"observation_digest":"sha256:5a6690a2e24e6e5a4a88905051de438c242f91c2f0f8def45e7e3754d13e68e5","observation_id":"d962fbbe-3828-4961-9347-cf0e52fb0db0","resolution":{"observed_at":"2026-08-06T15:02:03.011140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2508.02324","last_updated":"2025-08-04T11:49:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-04T11:49:20Z","title":"Qwen-Image Technical Report","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T14:29:06.883874Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2508.02324"},"observation_digest":"sha256:81b1314298b598b86bf4c012b909c6cced93ba454a869b972dbcf63220a4d7f5","observation_id":"4a0f15e7-753a-4a95-8f51-f6470d7277c7","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":183,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:b6a387a58f4a6ffb168a85bf3061342016b160f04583f48cf9ba8cad88cefac3","observation_id":"7c155d67-d838-4042-89da-c5a4e5893ef5","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T12:27:27.708238Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.01610","last_updated":"2025-09-01T16:43:48Z","snapshot_observed_at":"2026-08-08T14:14:58.948394Z","submitted_at":"2025-09-01T16:43:48Z","title":"Improving Large Vision and Language Models by Learning from a Panel of Peers","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-05T12:27:27.708238Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2509.01610"},"observation_digest":"sha256:5bac2dcc37828bb27a92f5bc276d6cf5c7cf8a98f4b20ee70387c0cf97897b3e","observation_id":"b5d7a50d-8486-46ac-bcce-8160e888fd94","resolution":{"observed_at":"2026-08-05T12:27:27.708238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T11:18:08.578239Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.03018","last_updated":"2025-09-03T04:49:20Z","snapshot_observed_at":"2026-08-11T00:26:46.236448Z","submitted_at":"2025-09-03T04:49:20Z","title":"Mycroft: Tracing Dependencies in Collective Communication Towards Reliable LLM Training","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-05T11:18:08.578239Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2509.03018"},"observation_digest":"sha256:506f8ada0aefd781bd3c8014f8edb98054cc40618b9a8a323456340fb3ca50f7","observation_id":"5cc7efa1-388f-4400-a4b5-7d5d29c8381c","resolution":{"observed_at":"2026-08-05T11:18:08.578239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T10:19:54.532964Z","title":"Pytorch fsdp: experi- ences on scaling fully sharded data parallel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.04394","last_updated":"2025-09-04T17:05:59Z","snapshot_observed_at":"2026-08-10T04:42:03.007150Z","submitted_at":"2025-09-04T17:05:59Z","title":"Transition Models: Rethinking the Generative Learning Objective","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-05T10:19:54.532964Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2509.04394"},"observation_digest":"sha256:211b9a3e19e6ed3b25a9f3e7614d9987e7a41173123dc1101526e61fa7bc4b44","observation_id":"858f84ff-2871-450d-a905-2e601dc4232f","resolution":{"observed_at":"2026-08-05T10:19:54.532964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T05:29:37.545479Z","title":"Pytorch fsdp: experiences on scaling fully sharded data par- allel","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.07003","last_updated":"2025-09-05T19:29:00Z","snapshot_observed_at":"2026-08-09T15:12:25.740960Z","submitted_at":"2025-09-05T19:29:00Z","title":"veScale: Consistent and Efficient Tensor Programming with Eager-Mode SPMD","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T05:29:37.545479Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2509.07003"},"observation_digest":"sha256:fbcbdacac74b473474cbbc63d0a7bd5ad9ff9b6d2849769e9e2c0ba266111335","observation_id":"ab6be7d6-cc9f-485c-a37c-9de611f152b1","resolution":{"observed_at":"2026-08-05T05:29:37.545479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-04T14:43:04.792216Z","title":"arXiv preprint arXiv:2304.11277(2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.23722","last_updated":"2026-07-03T02:44:15Z","snapshot_observed_at":"2026-08-04T14:42:54.705316Z","submitted_at":"2025-09-28T08:05:13Z","title":"OctoPipe: Reducing Pipeline Bubbles for Heterogeneous Models via Co-Optimizing Partitioning, Placement, and Scheduling","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-04T14:43:04.792216Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2509.23722"},"observation_digest":"sha256:71cb08f3734fd3d1b3be9884c4b186be42f366683c29e5c8e3ad1798db1548fb","observation_id":"054f5e2c-50dd-4cac-a66d-f0ebde3eeb98","resolution":{"observed_at":"2026-08-04T14:43:04.792216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2509.24527","last_updated":"2025-09-29T09:42:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-29T09:42:27Z","title":"Training Agents Inside of Scalable World Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-15T02:05:52.431747Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2509.24527"},"observation_digest":"sha256:dd9583ad7c5c3306e19edd70a06e660bd77050a1d22388cb32fbe7e88635ee23","observation_id":"ac072154-eb76-4864-982d-1a4f08a0a88e","resolution":{"observed_at":"2026-05-15T02:05:52.608605Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-04T12:55:12.140281Z","title":"arXiv preprint arXiv:2304.11277(2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.01565","last_updated":"2026-06-18T01:08:24Z","snapshot_observed_at":"2026-08-10T05:37:30.922463Z","submitted_at":"2025-10-02T01:23:32Z","title":"TetriServe: Efficiently Serving Mixed DiT Workloads","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T12:55:12.140281Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2510.01565"},"observation_digest":"sha256:c0b35510091c1a4bbac4b975275a17ee76fdc20547b063613465577fe9c4536f","observation_id":"699fd5e6-174c-4f1e-910e-447e35070f8e","resolution":{"observed_at":"2026-08-04T12:55:12.140281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2510.04800","last_updated":"2026-04-21T13:16:20Z","snapshot_observed_at":"2026-08-02T20:03:37.497118Z","submitted_at":"2025-10-06T13:30:07Z","title":"Hybrid Architectures for Language Models: Systematic Analysis and Design Insights","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-18T10:18:04.431436Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2510.04800"},"observation_digest":"sha256:5ece4694e701b1336ed2ae6bb8006200d591a00c9492464a94d3f7c6052f8039","observation_id":"3691cc5f-78f8-4795-93a2-ea3a5d51e1a5","resolution":{"observed_at":"2026-05-18T10:21:15.065547Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2510.08431","last_updated":"2026-05-06T15:49:52Z","snapshot_observed_at":"2026-08-11T10:50:12.121285Z","submitted_at":"2025-10-09T16:45:30Z","title":"Large Scale Diffusion Distillation via Score-Regularized Continuous-Time Consistency","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-18T08:46:16.541104Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2510.08431"},"observation_digest":"sha256:7ec275527b93b03edb5d1cd1e0962c968d5f53fc6cb552d93b6bde3523346062","observation_id":"d58667d6-4ecc-41f7-af3a-df470bcb4ac0","resolution":{"observed_at":"2026-05-18T08:51:09.131171Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2510.15596","last_updated":"2026-04-12T23:09:17Z","snapshot_observed_at":"2026-08-11T00:31:21.621627Z","submitted_at":"2025-10-17T12:41:37Z","title":"PRISM: Probabilistic Runtime Insights and Scalable Performance Modeling for Large-Scale Distributed Training","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-18T06:27:40.254366Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2510.15596"},"observation_digest":"sha256:42bc2ccb7f0c29f7e36e8b9e1f96d12bb64cf2d0a41935bae7b223abe2ea3ba4","observation_id":"3b12ba06-9a83-4d12-9abb-cb60191caeb4","resolution":{"observed_at":"2026-05-18T06:30:59.737230Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-04T08:26:59.755095Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel.arXiv preprint arXiv:2304.11277, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.20769","last_updated":"2026-06-24T20:52:29Z","snapshot_observed_at":"2026-08-09T01:24:24.242725Z","submitted_at":"2025-10-23T17:43:38Z","title":"CSU-PCAST: A Dual-Branch Transformer Framework for medium-range ensemble Precipitation Forecasting","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T08:26:59.755095Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2510.20769"},"observation_digest":"sha256:63e84c1745184d3d92afad9a3dc3117d2f263f77acd874c9463dadbcc8c48481","observation_id":"265200c5-91dc-4e14-860d-2ff68633e242","resolution":{"observed_at":"2026-08-04T08:26:59.755095Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2511.06605","last_updated":"2026-04-10T17:41:06Z","snapshot_observed_at":"2026-08-02T17:48:25.735846Z","submitted_at":"2025-11-10T01:28:58Z","title":"DMA-Latte: Expanding the Reach of DMA Offloads to Latency-bound ML Communication","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-18T00:27:43.845408Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2511.06605"},"observation_digest":"sha256:a9707831d2d65f928abec2b3e6e3125fd9290ed0ba62213c5d1d2e3d72376b55","observation_id":"5c627fe8-8df7-40ea-ab1b-7ea9a2723bff","resolution":{"observed_at":"2026-05-18T00:30:33.088064Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2511.09861","last_updated":"2026-05-12T23:31:13Z","snapshot_observed_at":"2026-08-11T04:36:49.811563Z","submitted_at":"2025-11-13T01:41:47Z","title":"Lit Silicon: A Case Where Thermal Imbalance Couples Concurrent Execution in Multiple GPUs","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-17T23:11:11.543271Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2511.09861"},"observation_digest":"sha256:4db55e037c63e82366d72cf6be08fcfa904b6e3869879bc6e7d10e73cf3c6d44","observation_id":"75589c0f-deb6-4f6d-8b3e-bafb4a1a9745","resolution":{"observed_at":"2026-05-17T23:12:11.892986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-03T22:28:52.826267Z","title":"Available: https://arxiv.org/abs/2304.11277","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.10480","last_updated":"2026-06-26T19:59:05Z","snapshot_observed_at":"2026-08-09T19:17:08.259662Z","submitted_at":"2025-11-13T16:44:56Z","title":"Scalable Synthesis of distributed LLM workloads through Symbolic Tensor Graphs","version":3},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-03T22:28:52.826267Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2511.10480"},"observation_digest":"sha256:a52a8491e3f51f8789d6b8ecaebbb4f41f64e6ec013f4857ce74875bd75799fc","observation_id":"8490d60b-4e3c-4736-b4a7-b5ef0a4d4939","resolution":{"observed_at":"2026-08-03T22:28:52.826267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-03T21:34:17.759091Z","title":"Pytorch fsdp: experiences onscalingfullyshardeddataparallel.arXivpreprintarXiv:2304.11277,2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.14993","last_updated":"2026-05-26T15:07:50Z","snapshot_observed_at":"2026-08-03T23:39:38.668973Z","submitted_at":"2025-11-19T00:23:22Z","title":"Kandinsky 5.0: A Family of Foundation Models for Image and Video Generation","version":3},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-03T21:34:17.759091Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2511.14993"},"observation_digest":"sha256:a44559159176ecce71bd4bb7817308419aa022a34612142f75d5b445a3a01ddd","observation_id":"44ad16c6-e06b-4ae9-b5c7-5c67c966a172","resolution":{"observed_at":"2026-08-03T21:34:17.759091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2511.22699","last_updated":"2026-07-06T06:19:03Z","snapshot_observed_at":"2026-08-03T19:47:26.577384Z","submitted_at":"2025-11-27T18:52:07Z","title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","version":3},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-11T14:08:36.801359Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2511.22699"},"observation_digest":"sha256:e14053c2b583d6e16f2d9fba3119f7a66e71162b55503f8301292a4893df9431","observation_id":"e35c2efa-d354-441a-aafa-f31c8c9a6e6d","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-03T19:47:32.963177Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel.arXiv preprint arXiv:2304.11277, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.22699","last_updated":"2026-07-06T06:19:03Z","snapshot_observed_at":"2026-08-03T19:47:26.577384Z","submitted_at":"2025-11-27T18:52:07Z","title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","version":5},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-03T19:47:32.963177Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2511.22699"},"observation_digest":"sha256:5096cb327bff73a14d942bb1541e98e9515ec9ec10ae8f6a50a9245131690a23","observation_id":"e5b84ebc-1b0e-4bc5-bdc0-b8b55e55a319","resolution":{"observed_at":"2026-08-03T19:47:32.963177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2512.10955","last_updated":"2026-04-30T17:59:43Z","snapshot_observed_at":"2026-08-04T18:59:53.204146Z","submitted_at":"2025-12-11T18:59:56Z","title":"Omni-Attribute: Open-vocabulary Attribute Encoder for Visual Concept Personalization","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-16T22:52:41.592714Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2512.10955"},"observation_digest":"sha256:b9efff2738f62258e9ddad567890e15825411efab455e6ec3b367a8bf7c60077","observation_id":"fa9d6c83-3baa-4219-896c-a82c8b9635fc","resolution":{"observed_at":"2026-05-16T22:53:38.251173Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2512.12131","last_updated":"2026-05-11T19:05:27Z","snapshot_observed_at":"2026-08-11T00:07:48.019440Z","submitted_at":"2025-12-13T01:50:18Z","title":"BOOST: BOttleneck-Optimized Scalable Training Framework for Low-Rank Large Language Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T23:19:02.358348Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2512.12131"},"observation_digest":"sha256:fc8ee00ee148f77cf5de845631de5e88f7611235947fde9a482d3edc58725d67","observation_id":"99545e23-bf70-4d1b-b965-54e6cd390aff","resolution":{"observed_at":"2026-05-16T23:21:21.589289Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-03T10:35:06.660201Z","title":"Pytorch fsdp: experiences on scaling fully sharded data par- allel.arXiv preprint arXiv:2304.11277, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.09881","last_updated":"2026-07-09T22:49:12Z","snapshot_observed_at":"2026-08-09T17:44:11.904091Z","submitted_at":"2026-01-14T21:30:03Z","title":"Transition Matching Distillation for Fast Video Generation","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-03T10:35:06.660201Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2601.09881"},"observation_digest":"sha256:48800a28ae3eca6b7e450a64ef41e2f6852f9148c6c27e7be343b79b0f5ebd86","observation_id":"ca2ccf9a-d1f0-49bd-abe5-d3f1dac014b3","resolution":{"observed_at":"2026-08-03T10:35:06.660201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2601.10611","last_updated":"2026-04-02T16:01:02Z","snapshot_observed_at":"2026-08-02T06:45:30.387180Z","submitted_at":"2026-01-15T17:27:44Z","title":"Molmo2: Open Weights and Data for Vision-Language Models with Video Understanding and Grounding","version":4},"reference_index":185,"source":"pdf_text","source_observed_at":"2026-05-16T04:21:29.526008Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2601.10611"},"observation_digest":"sha256:f24dacf51caad8e9665bc14f4b8a94f02ea5066edc06d6865b549319fc57f050","observation_id":"c9fff4ad-fcd4-4eb4-9ad9-b9b5b25e9a38","resolution":{"observed_at":"2026-05-16T04:21:29.906567Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-03T08:30:55.350781Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.16956","last_updated":"2026-01-23T18:26:14Z","snapshot_observed_at":"2026-08-10T02:09:28.166108Z","submitted_at":"2026-01-23T18:26:14Z","title":"DataStates-LLM: Scalable Checkpointing for Transformer Models Using Composable State Providers","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T08:30:55.350781Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2601.16956"},"observation_digest":"sha256:9f73c27d094e410477d0c22cf023af6f160a56282cb36edc8527bf61b0e787e7","observation_id":"92d54919-66cd-4825-a8e5-b7a6181c66e8","resolution":{"observed_at":"2026-08-03T08:30:55.350781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2601.20540","last_updated":"2026-01-28T12:37:01Z","snapshot_observed_at":"2026-08-03T01:29:14.204066Z","submitted_at":"2026-01-28T12:37:01Z","title":"Advancing Open-source World Models","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-16T09:07:00.904794Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2601.20540"},"observation_digest":"sha256:08fd98464e63b61f17dc4e9fd1563a7e1750b1fbf701f195db7928edf571c7ee","observation_id":"cbf4f22c-298e-4d17-b16a-2d9f5ea205ee","resolution":{"observed_at":"2026-05-16T09:07:01.078103Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-03T03:57:32.396743Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel.arXiv preprint arXiv:2304.11277,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.06442","last_updated":"2026-06-01T15:54:20Z","snapshot_observed_at":"2026-08-03T03:57:29.747868Z","submitted_at":"2026-02-06T07:11:50Z","title":"ChatUMM: Robust Context Tracking for Conversational Interleaved Generation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T03:57:32.396743Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2602.06442"},"observation_digest":"sha256:6c4e3b9be3d3dc232d2c882a793feeede2e8bd356d03f2b5384f031e02b0e990","observation_id":"be1acf60-4fa2-4e23-b65b-843ccb443109","resolution":{"observed_at":"2026-08-03T03:57:32.396743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-03T03:13:27.566126Z","title":"Pytorch fsdp: experiences on scaling fully sharded data parallel.arXiv preprint arXiv:2304.11277, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.08923","last_updated":"2026-07-07T20:21:24Z","snapshot_observed_at":"2026-08-08T01:05:15.308832Z","submitted_at":"2026-02-09T17:25:37Z","title":"DynamiQ: Accelerating Gradient Synchronization using Compressed Multi-hop All-reduce","version":3},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-03T03:13:27.566126Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2602.08923"},"observation_digest":"sha256:b833156f2bd62d74acc57d1d066110829e699dc13c4c80a0a5ca29bdbf785359","observation_id":"5ea45fbc-a51e-4908-b928-8900acf4ef29","resolution":{"observed_at":"2026-08-03T03:13:27.566126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-02T23:51:02.492413Z","title":"arXiv preprint arXiv:2304.11277(2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.12521","last_updated":"2026-07-03T06:44:30Z","snapshot_observed_at":"2026-08-11T01:08:51.925162Z","submitted_at":"2026-02-13T01:59:03Z","title":"Opus: Photonic Rail-Optimized Fabric in ML Datacenters","version":3},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-02T23:51:02.492413Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2602.12521"},"observation_digest":"sha256:c91e4e6099863fce56c9689ec04e4fe78a1122a154775ddb27d1ab715addb1cb","observation_id":"83550848-7977-4011-b1f8-6b5ac8771e2b","resolution":{"observed_at":"2026-08-02T23:51:02.492413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-02T21:11:51.292724Z","title":"Pytorch fsdp: Experiences on scaling fully sharded data parallel, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.21196","last_updated":"2026-07-10T12:36:34Z","snapshot_observed_at":"2026-08-10T08:47:24.615042Z","submitted_at":"2026-02-24T18:54:39Z","title":"Untied Ulysses: Memory-Efficient Context Parallelism via Headwise Chunking","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-02T21:11:51.292724Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2602.21196"},"observation_digest":"sha256:f40438552ee555215a3651779e84f30a26400eeaf086a6735286cf35d5d273d3","observation_id":"f4cfafaf-795e-49b0-bafe-c0b38fc668cf","resolution":{"observed_at":"2026-08-02T21:11:51.292724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2602.22437","last_updated":"2026-04-21T21:24:42Z","snapshot_observed_at":"2026-08-06T03:16:16.624191Z","submitted_at":"2026-02-25T21:55:43Z","title":"veScale-FSDP: Flexible and High-Performance FSDP at Scale","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T19:03:22.142671Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2602.22437"},"observation_digest":"sha256:550d155a708530e2cf78216430c0f1b80a89fbd8ac87084ce0ca1c579ced88d6","observation_id":"e8b2ffe0-b410-4f20-a2d0-d32ae53856af","resolution":{"observed_at":"2026-05-15T19:06:30.833896Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-02T18:23:40.250047Z","title":"arXiv preprint arXiv:2304.11277 (2023) 8","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.12262","last_updated":"2026-07-17T07:23:56Z","snapshot_observed_at":"2026-08-09T09:42:44.258435Z","submitted_at":"2026-03-12T17:59:51Z","title":"Video Streaming Thinking: VideoLLMs Can Watch and Think Simultaneously","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-02T18:23:40.250047Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2603.12262"},"observation_digest":"sha256:9a014373bb2ea33794eb5a6a295cb3f96e36a64bf832a30f779d4be72117e747","observation_id":"a9f1a70e-3ada-4f53-85b9-c40cc55358d4","resolution":{"observed_at":"2026-08-02T18:23:40.250047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2603.17812","last_updated":"2026-04-07T18:00:52Z","snapshot_observed_at":"2026-07-06T22:49:33.482934Z","submitted_at":"2026-03-18T15:04:57Z","title":"ChopGrad: Pixel-Wise Losses for Latent Video Diffusion via Truncated Backpropagation","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-15T09:42:26.074609Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2603.17812"},"observation_digest":"sha256:196f86abbdb861f42fde7b4d438e1c9c098b3c86fc9c4bd924f03cbb95532d99","observation_id":"6a235eae-aa42-4139-88e0-e031fd9bf549","resolution":{"observed_at":"2026-05-15T09:45:23.298668Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-07-13T11:43:31.089245Z","title":"Yiran Zhao, Wenyue Zheng, Tianle Cai, Xuan Long Do, Kenji Kawaguchi, Anirudh Goyal, and Michael Shieh","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.03962","last_updated":"2026-04-05T04:48:15Z","snapshot_observed_at":"2026-07-13T11:43:29.708599Z","submitted_at":"2026-04-05T04:48:15Z","title":"Predict, Don't React: Value-Based Safety Forecasting for LLM Streaming","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-13T11:43:31.089245Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.03962"},"observation_digest":"sha256:522b89137c7ef5e79f07f5a28cd3cc1dfb23a41a6b1d39eaa50a3b7361c27876","observation_id":"d4f75705-77e0-43b2-8ef7-5051ebe1f6b5","resolution":{"observed_at":"2026-07-13T11:43:31.089245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2604.04142","last_updated":"2026-04-05T15:00:29Z","snapshot_observed_at":"2026-07-06T22:53:10.816748Z","submitted_at":"2026-04-05T15:00:29Z","title":"OP-GRPO: Efficient Off-Policy GRPO for Flow-Matching Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-13T16:46:30.674244Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.04142"},"observation_digest":"sha256:6d7e6602d4bf51f8c7c78b3ebc9c38d55b06f85559b345ed19743b9aab020829","observation_id":"66fb65d3-2492-4557-bf24-a29cd3d8c709","resolution":{"observed_at":"2026-05-13T16:48:02.835106Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2604.04736","last_updated":"2026-04-06T15:03:35Z","snapshot_observed_at":"2026-08-11T09:04:33.166374Z","submitted_at":"2026-04-06T15:03:35Z","title":"Sampling Parallelism for Fast and Efficient Bayesian Learning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T18:48:24.778806Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.04736"},"observation_digest":"sha256:42b8f843e24a6b79c70e4babd021587b733f96d7a34cfb0e8039978dd2b0c5ee","observation_id":"72d40269-a202-4f87-badb-5786d89ad808","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2604.04750","last_updated":"2026-04-09T14:13:19Z","snapshot_observed_at":"2026-07-06T22:53:37.357996Z","submitted_at":"2026-04-06T15:16:35Z","title":"DeepStack: Scalable and Accurate Design Space Exploration for Distributed 3D-Stacked AI Accelerators","version":2},"reference_index":122,"source":"pdf_text","source_observed_at":"2026-05-10T19:04:17.725111Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.04750"},"observation_digest":"sha256:669ee12d115c8c46d5daa7c0ce2c16e55dbe453d7d04b9995f82083d179a216c","observation_id":"8fc9e5ad-047f-4f6c-a1e8-dd21007c7af6","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2604.05091","last_updated":"2026-04-06T18:43:56Z","snapshot_observed_at":"2026-07-06T22:53:55.926521Z","submitted_at":"2026-04-06T18:43:56Z","title":"MegaTrain: Full Precision Training of 100B+ Parameter Large Language Models on a Single GPU","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T18:57:25.256574Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.05091"},"observation_digest":"sha256:3ed2f7b1ad01745496e342035ccbe243105873af0df9d5decfcb1a472b3d4f50","observation_id":"104f6be9-8249-4981-bc77-317912439ab0","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2604.05426","last_updated":"2026-04-10T07:32:38Z","snapshot_observed_at":"2026-08-11T05:30:27.043835Z","submitted_at":"2026-04-07T04:40:17Z","title":"ALTO: Adaptive LoRA Tuning and Orchestration for Heterogeneous LoRA Training Workloads","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T20:21:27.428360Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.05426"},"observation_digest":"sha256:e386b82292796be9b55716a4afe9fb697d7fa0ed071abe9beeabf5c59d0679a9","observation_id":"9ec0c8af-f80d-441a-9056-e60530b3732f","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2604.11521","last_updated":"2026-04-13T14:23:31Z","snapshot_observed_at":"2026-07-31T08:57:40.954187Z","submitted_at":"2026-04-13T14:23:31Z","title":"Continuous Adversarial Flow Models","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-10T15:25:53.420119Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.11521"},"observation_digest":"sha256:85b421f2c8965bd9981679d927bddac8b16477adbb93cc9bb8e586fd90cd9a53","observation_id":"39de9997-d347-4041-8aac-0eea2c8ab661","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"cited_work":{"arxiv_id":"2304.11277","doi":"10.48550/arxiv.2304.11277","metadata_source":"pith","pith_arxiv_id":"2304.11277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","venue":"cs.DC","work_id":"bee7755e-b855-401d-813a-06ae9451d768","year":2023},"citing_paper":{"arxiv_id":"2604.11554","last_updated":"2026-04-14T09:26:26Z","snapshot_observed_at":"2026-08-09T12:36:17.368609Z","submitted_at":"2026-04-13T14:42:03Z","title":"Relax: An Asynchronous Reinforcement Learning Engine for Omni-Modal Post-Training at Scale","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T16:07:52.017037Z"},"links":{"cited_paper":"/paper/2304.11277","citing_paper":"/paper/2604.11554"},"observation_digest":"sha256:49d04be26734d62914fda2e66097a12609f38cebb5a87ab8c20606a3e5f22cc4","observation_id":"b2f940ef-fad4-45de-a12e-be7f400cce3d","resolution":{"observed_at":"2026-05-12T04:15:20.153254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T14:22:13.496008+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2304.11277/citation-record","integrity":"/paper/2304.11277/integrity","json":"/paper/2304.11277/citation-record.json","paper":"/paper/2304.11277"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"torch.amp Gradient Scaling","venue":null,"work_id":"2df66c0e-846a-4b98-be97-2e5d8a7dc27b","year":2023},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:167c358f02e553d8815f5e3a7a1bd755c0f5d26c0dc2a26e94347ba30452f9cd","observation_id":"c858bb33-93d2-4285-9e41-9d10d3cccf40","resolution":{"observed_at":"2026-05-12T04:15:20.144142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b1a066c3-0f93-46a9-8e26-303cd09ec003","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:4d1e3155738fc0e2b2f1968e4750c01936a5c57663065127f9d71244e1162b2c","observation_id":"99f20aa3-13f0-4aaf-8406-1d8cab85679f","resolution":{"observed_at":"2026-05-12T04:15:20.114455Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"951cb6e2-2d17-4acd-b617-d2a64f7dbd72","year":2020},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:d1fed1980a1594bbd001928063bdd05c40d06559562dedad3f7bdbdf66486c60","observation_id":"55683cfe-58e5-415a-a57d-ca0d4e873138","resolution":{"observed_at":"2026-05-12T04:15:20.146037Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"572e8d20-fcbb-4977-90d1-9c0b250c08b1","year":2022},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:94fffb5bbc8bf9a2bb8ffedf87b59a89900d6fbb9f71c12e9d57269955ca129d","observation_id":"26facca3-4f42-4bbe-92ac-8a6529cd09de","resolution":{"observed_at":"2026-05-12T04:15:20.140258Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1806.03377","last_updated":"2018-06-08T23:18:08Z","snapshot_observed_at":"2026-08-02T18:12:03.258298Z","submitted_at":"2018-06-08T23:18:08Z","title":"PipeDream: Fast and Efficient Pipeline Parallel DNN Training","version":1},"cited_work":{"arxiv_id":"1806.03377","doi":"10.48550/arxiv.1806.03377","metadata_source":"pith","pith_arxiv_id":"1806.03377","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PipeDream: Fast and Efficient Pipeline Parallel DNN Training","venue":"cs.DC","work_id":"335ca03b-43f7-43d8-af32-3eaeb6735100","year":2018},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/1806.03377","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:5a520b877e4189442a1efc36702c8e12524dd9ab77d5b3625c0d818e237f1877","observation_id":"d5fa683c-080d-4f90-80c6-d0c86f35f8ee","resolution":{"observed_at":"2026-05-12T04:15:20.068407Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.03161","last_updated":"2021-02-12T12:26:03Z","snapshot_observed_at":"2026-08-09T05:31:10.979727Z","submitted_at":"2021-02-05T13:39:31Z","title":"PipeTransformer: Automated Elastic Pipelining for Distributed Training of Transformers","version":2},"cited_work":{"arxiv_id":"2102.03161","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2102.03161","snapshot_observed_at":"2026-07-04T13:29:51.165701Z","title":"arXiv preprint arXiv:2102.03161 , year=","venue":null,"work_id":"fdfe466f-cde1-4744-b78b-511d55948fba","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2102.03161","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:8dd95825cfd9493c488fdc0f8744880c89c1f15d14fe5f28e38c9c3c2e4e6ecc","observation_id":"80025e5d-515b-403d-b32b-77c3118d7720","resolution":{"observed_at":"2026-05-12T04:15:20.072364Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"594dc9c2-5bd6-4b8c-a4f2-54e769ddca1b","year":2022},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:155489fc450aafbcff8ac322ab3bf17f7b86cbdf843586214a3fbc7aed6f0683","observation_id":"a5a0a9c0-738d-4caa-8b07-d5744ad61e09","resolution":{"observed_at":"2026-05-12T04:15:20.136516Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0d0befd7-2847-4977-9581-cab5595e04ff","year":2019},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:07ec3861574c41be5fdd7f2ac5b7805a70ae1e0ba3b039553e35fbe35d23d315","observation_id":"19519cd9-5de7-4396-a206-96633f201bf2","resolution":{"observed_at":"2026-05-12T04:15:20.138249Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.1807","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Accurate Uncertainties for Dee p Learning Using Calibrated Regression","venue":null,"work_id":"563a857f-22b3-4a6a-91b4-2dbbff389cc6","year":2018},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:cd3450e73a9eaf5562aa6a3884433add0161a0a0b6a5b8e6a967b6460e147722","observation_id":"2c846745-31d0-401d-82eb-b1e9bbf0b6bb","resolution":{"observed_at":"2026-05-12T04:15:20.052543Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1b3f3db9-9471-4309-a28c-62196260143f","year":2020},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:96eabf1feb3c85195c10c505293c4f40d7fc9cdc64288a3ea1daac2cb9d35a1e","observation_id":"57d56363-abd9-45f9-bf9f-834846974d51","resolution":{"observed_at":"2026-05-12T04:15:20.142190Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.09910","last_updated":"2020-04-21T11:27:00Z","snapshot_observed_at":"2026-08-09T06:30:13.907413Z","submitted_at":"2020-04-21T11:27:00Z","title":"torchgpipe: On-the-fly Pipeline Parallelism for Training Giant Models","version":1},"cited_work":{"arxiv_id":"2004.09910","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2004.09910","snapshot_observed_at":"2026-07-04T02:39:24.645515Z","title":"torchgpipe: On-the-fly pipeline parallelism for training giant models","venue":null,"work_id":"dd4c1ae1-7618-40b2-b37e-8279c74e3737","year":2004},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2004.09910","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:0a954fe23296ab9c0fa94666a1c86c8a02e1dad502bf0ee65dd838c2878cdf9b","observation_id":"802594f0-03fa-42fd-8360-1f2893c8fe57","resolution":{"observed_at":"2026-05-12T04:15:20.075557Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.09616","last_updated":"2021-03-18T06:20:23Z","snapshot_observed_at":"2026-07-06T09:29:54.899176Z","submitted_at":"2020-06-17T02:49:59Z","title":"Dynamic Tensor Rematerialization","version":4},"cited_work":{"arxiv_id":"2006.09616","doi":"10.48550/arxiv.2006.09616","metadata_source":"arxiv_reference","pith_arxiv_id":"2006.09616","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Dynamic tensor rematerialization,","venue":"arXiv (Cornell University)","work_id":"998114da-0640-42f9-802a-797be8764ddd","year":2006},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2006.09616","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:8ee2640556943d95803d59d7af39f7fa8cbb48044fdc80342fe4e1dd1ed56a7a","observation_id":"e3bbfabb-1a63-4802-ac38-2f71709db5e9","resolution":{"observed_at":"2026-05-12T04:15:20.060933Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.05198","last_updated":"2022-05-10T22:40:17Z","snapshot_observed_at":"2026-08-10T09:51:23.811238Z","submitted_at":"2022-05-10T22:40:17Z","title":"Reducing Activation Recomputation in Large Transformer Models","version":1},"cited_work":{"arxiv_id":"2205.05198","doi":"10.48550/arxiv.2205.05198","metadata_source":"arxiv_reference","pith_arxiv_id":"2205.05198","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kuaishou","venue":"arXiv (Cornell University)","work_id":"95aeb8ec-f2a2-4ca9-b63d-125f9638d395","year":2024},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2205.05198","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:6623f3565a2ea47c759bde1bd534f1a9b538bbfde3296ff0d354bc8056bf758f","observation_id":"e3169a5d-bf93-40d0-b0f4-2af1cd38ed7b","resolution":{"observed_at":"2026-05-12T04:15:20.079382Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-21T12:26:21.423141+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T12:26:21.423141+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.15704","last_updated":"2020-06-28T20:39:45Z","snapshot_observed_at":"2026-07-06T09:33:26.082100Z","submitted_at":"2020-06-28T20:39:45Z","title":"PyTorch Distributed: Experiences on Accelerating Data Parallel Training","version":1},"cited_work":{"arxiv_id":"2006.15704","doi":"10.48550/arxiv.2006.15704","metadata_source":"pith","pith_arxiv_id":"2006.15704","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PyTorch Distributed: Experiences on Accelerating Data Parallel Training","venue":"cs.DC","work_id":"353279b8-3b33-45fd-9b64-41e5bd1708b9","year":2020},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2006.15704","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:04acd214c5fd95135ed54086c9427f5dabc8302f81de7d4bb53789b0b105e628","observation_id":"0c973995-1e60-4760-86af-1b5c7d7bedf2","resolution":{"observed_at":"2026-05-12T04:15:20.083330Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:22:38.135957+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:22:38.135957+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"856abbc5-8138-4a55-834d-fb932bf5cb97","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:d6456fcbd1d2943f8f8aadfa8fa24a798e5b91722608d1efd32e998a85ab4564","observation_id":"2daccb86-6682-45c1-a7bc-ab23caa78e08","resolution":{"observed_at":"2026-05-12T04:15:20.152588Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"29af2d9a-e55a-4087-9eec-c64e45e1a41c","year":2017},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:d81e4ffa754810a76b20f7eedb6a26b007c0c70ccc266fb9778c740e555c1d75","observation_id":"ed9f89a2-66a5-4cd2-9dcf-e5c462cef58a","resolution":{"observed_at":"2026-05-12T04:15:20.104955Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cec241b8-9cc7-421a-92c5-632e4a8b60f7","year":2020},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:09658710311eff2138fe26bdd74d45ef30856454776882557487f2f2371253d1","observation_id":"8d087009-fca7-4599-ac3c-f97a293091b4","resolution":{"observed_at":"2026-05-12T04:15:20.107236Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1710.03740","last_updated":"2018-02-15T20:04:02Z","snapshot_observed_at":"2026-07-06T06:03:37.247241Z","submitted_at":"2017-10-10T17:42:04Z","title":"Mixed Precision Training","version":3},"cited_work":{"arxiv_id":"1710.03740","doi":"10.48550/arxiv.1710.03740","metadata_source":"pith","pith_arxiv_id":"1710.03740","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mixed Precision Training","venue":"cs.AI","work_id":"c525941b-ce20-4bcb-8509-a9968f1e89c3","year":2017},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/1710.03740","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:2d2bf7801a299ff5e62e54e956f5ac306ae6970c1fdcc339a767053096a0c905","observation_id":"a5d5bb79-16c7-4e39-add7-b4fa1c58d15c","resolution":{"observed_at":"2026-05-12T10:47:18.676998Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:36.073032+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:36.073032+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c51a6db8-8166-40ce-b222-9f225ebd9362","year":null},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:589ab7ba05f9649e9dc8e97edd70c8f203a72a5df29aa2fd772dca52ce076b5f","observation_id":"3368299c-24dc-4a1c-becc-8a2187bd7761","resolution":{"observed_at":"2026-05-12T04:15:20.112411Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.05158","last_updated":"2023-02-27T00:21:53Z","snapshot_observed_at":"2026-08-08T12:11:29.797732Z","submitted_at":"2021-04-12T02:15:55Z","title":"Software-Hardware Co-design for Fast and Scalable Training of Deep Learning Recommendation Models","version":7},"cited_work":{"arxiv_id":"2104.05158","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2104.05158","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"High-performance, distributed training of large-scale deep learning recommendation models","venue":null,"work_id":"96983d8f-22fc-464c-8a9b-4d09ba266d67","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2104.05158","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:0aaca99835faa8977084a088afa38775a0b8a1b53f02573a059536d66836b2e7","observation_id":"9d5bb313-75c6-4286-8ff4-27edca4f021b","resolution":{"observed_at":"2026-05-12T04:15:20.090082Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"feaf6b33-9dc5-4cf8-a5a0-f648f1f20213","year":2019},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:49dbad47fa1510d2fcd341507fafb520bd430e4b3dd0287d6bb6ff1f8a290715","observation_id":"acd1af57-cead-41b8-ba02-0ab2d37726af","resolution":{"observed_at":"2026-05-12T04:15:20.116443Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"979659fe-aa39-4645-aeee-12eddc6e0334","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:e751e02f8ca32641b4816c5ae5c29f0acd45a15f08a2c3d7e95d3aa7d2760186","observation_id":"a8efa8cd-7a7c-4a77-b3f6-8e10d0755c5c","resolution":{"observed_at":"2026-05-12T04:15:20.118576Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0a3dcfb2-1526-463c-b931-6a3cafe0fe72","year":2023},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:678fbd46d9544ff06ebea5f11f3f2694d1c04db0f379a1a438faf21bc97f6afb","observation_id":"9991499a-50ac-4809-8f36-f5b7b2547d32","resolution":{"observed_at":"2026-05-12T04:15:20.120822Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a33381a3-474d-4768-80b5-2df3c19d7359","year":2023},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:b86354207f48a51f0a1cb6b090f52989e00ee88f9486aa23e6ddc2011189a3c3","observation_id":"55cdd3a3-e010-41e4-ae97-2b489fdb30f0","resolution":{"observed_at":"2026-05-12T04:15:20.122806Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6235a071-6dca-49b8-9f58-df11f0d63319","year":2019},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:a0675d09413ebf2fbabdb1b700b12ad6e4b458e79dbe40303fd754bfa5864863","observation_id":"ef6c5eed-5240-4d78-9e35-62f78fe9bd6b","resolution":{"observed_at":"2026-05-12T04:15:20.124721Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"944b9f47-16cf-4e3a-9d16-c12b4485c060","year":2023},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:c02bdefa552fd0426b95d357338586ec852be1b5534dc88387f9b93c532abcdc","observation_id":"fd260cb2-3bf8-4fc7-a5ae-5bc34c989782","resolution":{"observed_at":"2026-05-12T04:15:20.126852Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d2eb0ead-1c5b-45c5-a853-e6f046e1079f","year":2020},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:cd0fad514f43d75b1cec12cc51127965fa0ac67fe8af942d920bef65c8ddb2d4","observation_id":"9321594d-8d41-4712-8425-1159ae372d22","resolution":{"observed_at":"2026-05-12T04:15:20.128846Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9647b25b-2b03-4e6a-9949-e1701afd21b8","year":2020},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:a2dc18e6b2f5e4886794022eeb030ad7bb4de1dfb8174c251a399c86b2de5696","observation_id":"3f307cd8-69ce-4dfa-be83-090f8cda9b5d","resolution":{"observed_at":"2026-05-12T04:15:20.131050Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1dd46f5f-47a1-4f41-9ff0-b199d370554a","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:a0ab5a4fca967f0ac4e8c1c7156388465991306369133afb6895f0d66bedfea1","observation_id":"5bdbc883-d190-4cdc-a035-7f1da8d8453f","resolution":{"observed_at":"2026-05-12T04:15:20.132889Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5175366a-ee92-4b2f-94d0-7658190088d4","year":2017},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:8740aa8d9fca4c18ea2dac241bb9d476b01a6accb63df47de4243a3cb2c5c1c1","observation_id":"9a86ad3d-68ba-475a-8b31-6a9eb8d66ac6","resolution":{"observed_at":"2026-05-12T04:15:20.134720Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.13336","last_updated":"2020-04-28T07:13:50Z","snapshot_observed_at":"2026-07-06T09:15:46.584290Z","submitted_at":"2020-04-28T07:13:50Z","title":"Automatic Cross-Replica Sharding of Weight Update in Data-Parallel Training","version":1},"cited_work":{"arxiv_id":"2004.13336","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2004.13336","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Automatic cross-replica sharding of weight update in data-parallel training","venue":null,"work_id":"a0091ad0-1984-412f-ab68-26a9f4365607","year":2004},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2004.13336","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:6b6718eb1dfb358a8de24ed3319a52e5a239d6476069d1e7d6ee7dc0500b5a01","observation_id":"eaed3ba3-a101-4c90-8d53-12b669390a4d","resolution":{"observed_at":"2026-05-12T04:15:20.092977Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a6c57b52-64db-42ee-be72-6ba98aa3441f","year":null},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:bfb377d785ca0ccce2a67bc95ccde5508f229d83533998f0de7bfefbbdc3aa75","observation_id":"0ec2ff22-9726-42d1-9da9-e74451217978","resolution":{"observed_at":"2026-05-12T04:15:20.148161Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.04663","last_updated":"2021-12-23T21:29:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-05-10T20:54:58Z","title":"GSPMD: General and Scalable Parallelization for ML Computation Graphs","version":2},"cited_work":{"arxiv_id":"2105.04663","doi":"10.48550/arxiv.2105.04663","metadata_source":"pith","pith_arxiv_id":"2105.04663","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GSPMD: General and Scalable Parallelization for ML Computation Graphs","venue":"cs.DC","work_id":"0ab74606-fb17-4ead-898b-8086ae8cb3af","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2105.04663","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:129c7afd4837a768a39a6483902ca95d16422474511655c037814aa47665098b","observation_id":"ce53eb44-d4f5-417e-a2ec-171fc38f833f","resolution":{"observed_at":"2026-05-18T12:36:36.562059Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.15032","last_updated":"2022-04-19T11:57:54Z","snapshot_observed_at":"2026-08-10T18:25:17.200408Z","submitted_at":"2021-10-28T11:32:14Z","title":"OneFlow: Redesign the Distributed Deep Learning Framework from Scratch","version":6},"cited_work":{"arxiv_id":"2110.15032","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2110.15032","snapshot_observed_at":"2026-07-04T07:09:38.138172Z","title":"45 Geng Zhang, Xuanlei Zhao, Kai Wang, and Yang You","venue":null,"work_id":"4b54f68c-9e04-4726-83ed-ad871b3542f2","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2110.15032","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:49812a7745ae4373090690b6ed446859263a36f6e9cde649cb1f90404d8b0d10","observation_id":"c20ebb53-9d1b-4637-8d2d-186e107e4aec","resolution":{"observed_at":"2026-05-12T04:15:20.099914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11014","last_updated":"2022-03-11T21:19:31Z","snapshot_observed_at":"2026-08-10T01:07:51.015541Z","submitted_at":"2022-03-11T21:19:31Z","title":"DHEN: A Deep and Hierarchical Ensemble Network for Large-Scale Click-Through Rate Prediction","version":1},"cited_work":{"arxiv_id":"2203.11014","doi":"10.48550/arxiv.2203.11014","metadata_source":"arxiv_reference","pith_arxiv_id":"2203.11014","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Dhen: A deep and hierarchical ensemble network for large-scale click-through rate prediction","venue":"arXiv (Cornell University)","work_id":"ddad5535-eaa3-47a2-a069-372ccc48a32c","year":2022},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2203.11014","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:ce63ccbdb96395fd8605e49809aec9df23ae0ca37d1256917e6d3b9d79deef68","observation_id":"dc522653-2819-45b2-9554-edf99829462f","resolution":{"observed_at":"2026-05-12T04:15:20.057051Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-08T02:38:07.056521+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T02:38:07.056521+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.00119","last_updated":"2022-10-28T15:55:09Z","snapshot_observed_at":"2026-07-06T13:05:19.702730Z","submitted_at":"2022-04-30T00:55:29Z","title":"MiCS: Near-linear Scaling for Training Gigantic Model on Public Cloud","version":5},"cited_work":{"arxiv_id":"2205.00119","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2205.00119","snapshot_observed_at":"2026-06-29T21:03:58.748181Z","title":"arXiv preprint arXiv:2205.00119 , year=","venue":null,"work_id":"124ea131-64f3-4a7f-8938-9e36d87208a8","year":2022},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2205.00119","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:607eadc22e9892d733857c8402a8034551573c0649245dd89c6fc22dd10b8857","observation_id":"ace95df2-273a-4228-8001-907658cae18c","resolution":{"observed_at":"2026-05-12T04:15:20.102935Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"55a09388-5b8f-4b1e-8c73-0e12e9e2f324","year":null},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:26785309c9b3f28e0ce97984e7d95797e32aeda57cc950377ae18e6f56bf90c9","observation_id":"777bbed9-7a19-42e8-88a5-022fb0da44a4","resolution":{"observed_at":"2026-05-12T04:15:20.150762Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","venue":null,"work_id":"42b2c56d-f7dc-48aa-a350-25cb55620f3a","year":null},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:8e7b5ebbcd0f130bf1f1b5d605d972b8b43ab52816ff56d1dca0aa5fb6b3c6f4","observation_id":"b535e0f9-74ae-4dc4-aade-f56698679fd6","resolution":{"observed_at":"2026-05-12T04:15:20.110389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.11886","last_updated":"2021-04-19T07:06:02Z","snapshot_observed_at":"2026-07-06T10:52:09.106464Z","submitted_at":"2021-03-22T14:32:07Z","title":"DeepViT: Towards Deeper Vision Transformer","version":4},"cited_work":{"arxiv_id":"2103.11886","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2103.11886","snapshot_observed_at":"2026-07-04T03:09:30.169113Z","title":"Deepvit: Towards deeper vision transformer","venue":null,"work_id":"d16918d1-74f6-438b-a892-702f2b0cf120","year":2021},"citing_paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-12T04:15:20.027659Z"},"links":{"cited_paper":"/paper/2103.11886","citing_paper":"/paper/2304.11277"},"observation_digest":"sha256:9e9f2ede808a5eb0b28cd5814802f0fb12f6fa4d7a98638d76953a61873c54e3","observation_id":"ce8daacb-daca-4d89-9570-49e168886371","resolution":{"observed_at":"2026-05-12T04:15:20.064603Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2304.11277","last_updated":"2023-09-12T16:28:00Z","latest_version":2,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-01T19:01:47.393546Z","submitted_at":"2023-04-21T23:52:27Z","title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":22,"verified_exact":13,"verified_fuzzy":2},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 100 inbound Pith citation observations for arXiv:2304.11277."}