{"as_of":"2026-08-08T09:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:abc808102848f32996ab3876fbb6630a561568b951cf8b61afb315819ff851ca","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":29,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:05:17.834796Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":273,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2006.15704","last_updated":"2020-06-28T20:39:45Z","snapshot_observed_at":"2026-07-06T09:33:26.082100Z","submitted_at":"2020-06-28T20:39:45Z","title":"PyTorch Distributed: Experiences on Accelerating Data Parallel Training","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T19:15:26.764929Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2006.15704"},"observation_digest":"sha256:a27e8fe1ca8bcd03066cf5ccb9b747dc2bef025145c3dd8003d1d5c3bf66bca2","observation_id":"e1ad9f5c-7c94-4fca-b410-26e8483eec12","resolution":{"observed_at":"2026-05-17T19:15:26.852762Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2303.08112","last_updated":"2025-11-11T01:13:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-14T17:47:09Z","title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","version":6},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-12T16:54:37.382049Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2303.08112"},"observation_digest":"sha256:5bd698f9431c944abe2c04c75a3060c4bc9dd721746585f2c9fac0737627e4dd","observation_id":"686627a5-f0e5-4e24-a436-285b6556240f","resolution":{"observed_at":"2026-05-12T16:54:37.594176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2502.05171","last_updated":"2025-02-17T17:14:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-07T18:55:02Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-12T15:39:40.845703Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2502.05171"},"observation_digest":"sha256:b30ec98de2d9d0c1955090c7107de44d1c825998844555be0ad57593b8318d28","observation_id":"2a778857-1a0e-42dd-9422-7548c2e75db6","resolution":{"observed_at":"2026-05-12T15:39:41.141062Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-07T15:05:17.834796Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16463","last_updated":"2025-06-18T18:55:01Z","snapshot_observed_at":"2026-08-08T01:31:47.483168Z","submitted_at":"2025-05-22T09:44:44Z","title":"AnchorFormer: Differentiable Anchor Attention for Efficient Vision Transformer","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:05:17.834796Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2505.16463"},"observation_digest":"sha256:73aaef452d5ed99a089dcddb55e41ce0c26b49a2d1a145643eceaaf6e109aa9f","observation_id":"2ac5c55d-a864-4dac-b941-0025254eda76","resolution":{"observed_at":"2026-08-07T15:05:17.834796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-07T12:40:36.354788Z","title":"Reducing transformer depth on demand with structured dropout","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2505.23947","last_updated":"2025-05-29T18:56:45Z","snapshot_observed_at":"2026-08-07T12:35:46.392086Z","submitted_at":"2025-05-29T18:56:45Z","title":"Position: The Future of Bayesian Prediction Is Prior-Fitted","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T12:40:36.354788Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2505.23947"},"observation_digest":"sha256:67bae0cfe0c852b2e9ebf13abf4b47807ce85926149d47233b6ee242d04e70c6","observation_id":"211c76de-123e-4481-a77a-df405572e788","resolution":{"observed_at":"2026-08-07T12:40:36.354788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-06T22:40:53.679247Z","title":"Reducing Transformer Depth on Demand with Structured Dropout , September 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.21103","last_updated":"2025-06-26T09:01:19Z","snapshot_observed_at":"2026-08-06T22:31:44.262653Z","submitted_at":"2025-06-26T09:01:19Z","title":"Learning to Skip the Middle Layers of Transformers","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T22:40:53.679247Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2506.21103"},"observation_digest":"sha256:6cd6c8a64a7eae4c0f6ec13492cd2743eb9bbc30802af2ea267656438e41abf6","observation_id":"3223e1a1-e09b-4363-b4b1-f661be8cbb40","resolution":{"observed_at":"2026-08-06T22:40:53.679247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-06T22:17:42.899383Z","title":"Reducing transformer depth on demand with structured dropout","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2506.22015","last_updated":"2025-07-03T07:20:35Z","snapshot_observed_at":"2026-08-06T22:10:47.087808Z","submitted_at":"2025-06-27T08:28:21Z","title":"Towards Universal & Efficient Model Compression via Exponential Torque Pruning","version":3},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T22:17:42.899383Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2506.22015"},"observation_digest":"sha256:cee9983ee069b82585b64056b7b6f22e1db6b7942d9ee0750bc04576512d498c","observation_id":"282d4541-9bd4-46a8-b938-768d71b1ce3b","resolution":{"observed_at":"2026-08-06T22:17:42.899383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-06T18:33:19.194766Z","title":"Reducing transformer depth on demand with structured dropout","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2507.07996","last_updated":"2025-07-10T17:59:53Z","snapshot_observed_at":"2026-08-06T18:25:28.097317Z","submitted_at":"2025-07-10T17:59:53Z","title":"Skip a Layer or Loop it? Test-Time Depth Adaptation of Pretrained LLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T18:33:19.194766Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2507.07996"},"observation_digest":"sha256:9a585d8d662f7c4e348bb1c877e78b64d0bd07dbd20b70aebd45d3b2dd67209b","observation_id":"54e80b2d-9a4e-45cd-b894-d3ffe7f781a1","resolution":{"observed_at":"2026-08-06T18:33:19.194766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-06T18:24:51.447217Z","title":null,"venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2507.08567","last_updated":"2025-08-07T11:18:14Z","snapshot_observed_at":"2026-08-06T23:11:17.291536Z","submitted_at":"2025-07-11T13:11:11Z","title":"AbbIE: Autoregressive Block-Based Iterative Encoder for Efficient Sequence Modeling","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T18:24:51.447217Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2507.08567"},"observation_digest":"sha256:a6dfbbe4d9132110e7c71cb9088732bedcd56343c1052e6c00b51b5e5f2eb63e","observation_id":"ec931f3a-5d9b-4e1a-b351-2bc27cd39f80","resolution":{"observed_at":"2026-08-06T18:24:51.447217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-06T13:22:44.726762Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.20749","last_updated":"2025-07-28T11:57:52Z","snapshot_observed_at":"2026-08-07T22:24:55.642367Z","submitted_at":"2025-07-28T11:57:52Z","title":"Investigating Structural Pruning and Recovery Techniques for Compressing Multimodal Large Language Models: An Empirical Study","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T13:22:44.726762Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2507.20749"},"observation_digest":"sha256:5ce1b9afc967f4a5809a4e29325b766937d56d6ae81b9af4870514b2fd4c262a","observation_id":"3776073e-b025-4559-a80c-f2d61610c801","resolution":{"observed_at":"2026-08-06T13:22:44.726762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2508.03949","last_updated":"2026-04-14T17:16:30Z","snapshot_observed_at":"2026-08-04T01:34:07.529496Z","submitted_at":"2025-08-05T22:32:32Z","title":"Model Compression vs. Adversarial Robustness: An Empirical Study on Language Models for Code","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-19T00:01:42.228190Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2508.03949"},"observation_digest":"sha256:a684f82436eb7309b4624d3c804592b8e0070bff74dcb38a576d461271cb3969","observation_id":"b220cf03-cacc-40ac-99fe-602821af9ace","resolution":{"observed_at":"2026-05-19T00:01:55.897855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2508.04427","last_updated":"2026-06-11T09:27:03Z","snapshot_observed_at":"2026-08-06T00:02:56.551873Z","submitted_at":"2025-08-06T13:14:20Z","title":"Decoding the Multimodal Maze: A Systematic Review on the Adoption of Explainability in Multimodal Attention-based Models","version":1},"reference_index":105,"source":"pdf_text","source_observed_at":"2026-05-19T00:34:38.247099Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2508.04427"},"observation_digest":"sha256:18bf938366dc4d1414d196473283c7513cc006d4a0e9df4b57f3526515cfe94b","observation_id":"e4d35bd6-cf40-4be1-9924-1ff35b2f3be8","resolution":{"observed_at":"2026-05-19T00:36:56.313975Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-06T00:02:58.946950Z","title":null,"venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2508.04427","last_updated":"2026-06-11T09:27:03Z","snapshot_observed_at":"2026-08-06T00:02:56.551873Z","submitted_at":"2025-08-06T13:14:20Z","title":"Decoding the Multimodal Maze: A Systematic Review on the Adoption of Explainability in Multimodal Attention-based Models","version":2},"reference_index":105,"source":"pdf_text","source_observed_at":"2026-08-06T00:02:58.946950Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2508.04427"},"observation_digest":"sha256:82a8ab92da1172903769cddcdc5c94f59071fe329c3d934606a622b6d955ab85","observation_id":"8b34d409-72f4-470e-8e2e-b23b693ea384","resolution":{"observed_at":"2026-08-06T00:02:58.946950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T21:16:22.237618Z","title":"Reducing transformer depth on demand with structured dropout","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2508.09262","last_updated":"2025-08-12T18:05:33Z","snapshot_observed_at":"2026-08-08T09:26:18.699272Z","submitted_at":"2025-08-12T18:05:33Z","title":"Harnessing Input-Adaptive Inference for Efficient VLN","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T21:16:22.237618Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2508.09262"},"observation_digest":"sha256:78e29483a2de737e092dd484960562d501612ed769fb498155f53a3127db31d5","observation_id":"1beebdf1-25db-475a-9bc5-eb1303dad0e5","resolution":{"observed_at":"2026-08-05T21:16:22.237618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T12:07:37.008912Z","title":"ArXivabs/1909.11556 (2019), https://api.semanticscholar","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2509.01903","last_updated":"2025-09-02T02:51:23Z","snapshot_observed_at":"2026-08-07T03:39:34.754728Z","submitted_at":"2025-09-02T02:51:23Z","title":"VISP: Volatility Informed Stochastic Projection for Adaptive Regularization","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T12:07:37.008912Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2509.01903"},"observation_digest":"sha256:f29e2e0df5f2027b146898d73737f9edce68a49af03f396b1413d64571acee4f","observation_id":"483f8e97-ee2b-4ef9-8ae3-c0b17994d552","resolution":{"observed_at":"2026-08-05T12:07:37.008912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2604.15351","last_updated":"2026-04-04T10:24:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-04T10:24:12Z","title":"Aletheia: Gradient-Guided Layer Selection for Efficient LoRA Fine-Tuning Across Architectures","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-13T18:35:23.814733Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2604.15351"},"observation_digest":"sha256:e7311f285cdad5f32a973c1872271a2c44c302bf2fabee0ecb60b2904e073620","observation_id":"3f4346a0-e7d6-4651-ba8f-7c985a2c2f7b","resolution":{"observed_at":"2026-05-13T18:38:07.556104Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2604.17286","last_updated":"2026-04-19T06:59:34Z","snapshot_observed_at":"2026-08-03T04:25:48.631257Z","submitted_at":"2026-04-19T06:59:34Z","title":"Depth Adaptive Efficient Visual Autoregressive Modeling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T06:22:55.035749Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2604.17286"},"observation_digest":"sha256:f2150986342503f967a4d31deb399f2db23143f31045abf53b64ad8f837168e7","observation_id":"d1ac1b49-3e8e-4b59-a63e-c429923851db","resolution":{"observed_at":"2026-05-10T06:26:27.625427Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2604.17465","last_updated":"2026-04-30T22:17:02Z","snapshot_observed_at":"2026-08-02T19:54:22.380858Z","submitted_at":"2026-04-19T14:30:13Z","title":"Language models recognize dropout and Gaussian noise applied to their activations","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T05:38:12.882914Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2604.17465"},"observation_digest":"sha256:a614979537720a2a388c0026b8b1c77cf9098e5c86ef58cc7132762a46768c36","observation_id":"c01b4b4a-0906-42c7-af38-7334de1511bf","resolution":{"observed_at":"2026-05-10T05:41:02.186788Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2604.22782","last_updated":"2026-04-03T14:56:17Z","snapshot_observed_at":"2026-08-02T15:24:09.520492Z","submitted_at":"2026-04-03T14:56:17Z","title":"Stochastic KV Routing: Enabling Adaptive Depth-Wise Cache Sharing","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T19:56:48.015363Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2604.22782"},"observation_digest":"sha256:4be25c40d40739e3e9be1e13551e5516bff3968e61572d055e12db1730c38226","observation_id":"7351aec5-c8c2-4627-b739-ae141d653acf","resolution":{"observed_at":"2026-05-13T19:58:12.041938Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2604.24380","last_updated":"2026-04-27T12:10:44Z","snapshot_observed_at":"2026-07-06T23:10:24.992210Z","submitted_at":"2026-04-27T12:10:44Z","title":"Structural Pruning of Large Vision Language Models: A Comprehensive Study on Pruning Dynamics, Recovery, and Data Efficiency","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-08T03:47:38.100037Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2604.24380"},"observation_digest":"sha256:0b7439fecd752f094f650042f6b6d1c0c0cba8d2aaa5ef134468afb9839e997b","observation_id":"67ff6be4-e9a7-4148-b7d1-cbc5aefda9fb","resolution":{"observed_at":"2026-05-11T21:56:14.586000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2604.26181","last_updated":"2026-05-01T02:27:23Z","snapshot_observed_at":"2026-07-06T23:11:52.954360Z","submitted_at":"2026-04-28T23:56:39Z","title":"SWAN: World-Aware Adaptive Multimodal Networks for Runtime Variations","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-07T16:08:12.351139Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2604.26181"},"observation_digest":"sha256:a9c4070d1944dde66f3a0333f51c1b0d47961f7409385b7c2362f5adddb7a0a0","observation_id":"bf2da7a7-a8d6-447c-a3d8-8ceccf243a33","resolution":{"observed_at":"2026-05-11T23:51:29.267468Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2605.06105","last_updated":"2026-05-07T12:21:03Z","snapshot_observed_at":"2026-07-06T23:18:36.537416Z","submitted_at":"2026-05-07T12:21:03Z","title":"Shallow Prefill, Deep Decoding: Efficient Long-Context Inference via Layer-Asymmetric KV Visibility","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T10:33:52.457620Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2605.06105"},"observation_digest":"sha256:1d0e57e5234b638888e3f7ee849dfcf41c33a3b5c1d9aea582e6e7052f948c8f","observation_id":"c13c2308-d5ae-45b4-981d-07a7b3f47ed0","resolution":{"observed_at":"2026-05-11T20:01:09.011862Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2605.12714","last_updated":"2026-05-12T20:22:45Z","snapshot_observed_at":"2026-08-01T21:01:43.481918Z","submitted_at":"2026-05-12T20:22:45Z","title":"Layer-wise Representation Dynamics: An Empirical Investigation Across Embedders and Base LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-14T21:50:10.564922Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2605.12714"},"observation_digest":"sha256:3bc85830c308ecc9dcff7271e03be64d66a1a5f54b47730c39b452b4035b4b16","observation_id":"e9c90031-af5b-441b-9322-bd19b77d0272","resolution":{"observed_at":"2026-05-14T21:58:04.030648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2605.14037","last_updated":"2026-05-13T18:58:16Z","snapshot_observed_at":"2026-08-04T00:53:16.855757Z","submitted_at":"2026-05-13T18:58:16Z","title":"Self-Pruned Key-Value Attention: Learning When to Write by Predicting Future Utility","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-15T05:35:09.705532Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2605.14037"},"observation_digest":"sha256:720b2e15c1fdc0da69f97dab90421ea99a795b570696d85ca4c6e9a2afd2651b","observation_id":"972ad1df-c680-4ad2-807f-6c3adf3cb646","resolution":{"observed_at":"2026-05-15T05:39:47.975021Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2606.06574","last_updated":"2026-06-04T17:59:58Z","snapshot_observed_at":"2026-07-06T23:46:24.981618Z","submitted_at":"2026-06-04T17:59:58Z","title":"Skip a Layer or Loop It? Learning Program-of-Layers in LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T01:56:34.435152Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2606.06574"},"observation_digest":"sha256:6df11ee0d28b83e6a20b51d12136cf0431b5a7ce51843ceebb6c302f70cb0ddb","observation_id":"a9ae38b3-1ea4-455c-add4-905c249bdbf4","resolution":{"observed_at":"2026-07-02T12:36:57.370073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2606.09131","last_updated":"2026-06-08T07:28:14Z","snapshot_observed_at":"2026-08-07T20:04:49.687813Z","submitted_at":"2026-06-08T07:28:14Z","title":"Late-Layer Fusion is Enough: Dual-Path Vision Token Routing for Multimodal Large Language Models under Visual Saturation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T16:32:09.784716Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2606.09131"},"observation_digest":"sha256:06e656bbbc61e79d068c938e3dea497020c7771e216850ce51af271d07624508","observation_id":"7cada08f-0e51-4b6a-b246-8e3b26a7c03d","resolution":{"observed_at":"2026-06-27T16:41:03.364124Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2606.23670","last_updated":"2026-06-22T17:56:25Z","snapshot_observed_at":"2026-07-06T23:58:20.975630Z","submitted_at":"2026-06-22T17:56:25Z","title":"Tapered Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T09:11:20.341634Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2606.23670"},"observation_digest":"sha256:080fc19600fd1fb533aa4882fd781f2e4bc7792770b8c56e6692e57b1be04278","observation_id":"557c46b0-6da5-4ea7-a4ed-dee2dfdb2bb1","resolution":{"observed_at":"2026-07-04T09:59:45.954130Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":"1909.11556","doi":"10.48550/arxiv.1909.11556","metadata_source":"arxiv_reference","pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:1909.11556 (2019)","venue":"arXiv (Cornell University)","work_id":"2309ff43-c4f2-41b0-8dbe-92332c15716f","year":1909},"citing_paper":{"arxiv_id":"2606.29983","last_updated":"2026-06-29T08:58:09Z","snapshot_observed_at":"2026-08-03T21:27:41.768539Z","submitted_at":"2026-06-29T08:58:09Z","title":"Stabilizing Extrapolation in Looped Transformers via Learned Stochastic Stopping","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-30T07:29:23.786653Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2606.29983"},"observation_digest":"sha256:da85b1617aaea9506cc82efca3264d109f33431508a7c9e8dd340540e3b727aa","observation_id":"23ae4396-631b-4c90-a1cd-57c22b8282ba","resolution":{"observed_at":"2026-06-30T07:34:21.633951Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.11556","snapshot_observed_at":"2026-08-05T18:20:38.521564Z","title":"Reducing transformer depth on demand with structured dropout.arxiv preprint arxiv: 1909.11556,","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2608.03480","last_updated":"2026-08-04T11:17:54Z","snapshot_observed_at":"2026-08-07T23:11:58.783842Z","submitted_at":"2026-08-04T11:17:54Z","title":"Efficient Multilingual Neural Machine Translation via Corpus-Driven Vocabulary Pruning: An English-Arabic Case Study","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T18:20:38.521564Z"},"links":{"cited_paper":"/paper/1909.11556","citing_paper":"/paper/2608.03480"},"observation_digest":"sha256:f11e6c73f633ed1399507d2f69fa0d52b2495663f483d9f6bd356047aca056eb","observation_id":"ee248a96-9780-4b07-b3fe-04d7b3b56d60","resolution":{"observed_at":"2026-08-05T18:20:38.521564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1909.11556/citation-record","integrity":"/paper/1909.11556/integrity","json":"/paper/1909.11556/citation-record.json","paper":"/paper/1909.11556"},"outbound":[],"paper":{"arxiv_id":"1909.11556","last_updated":"2019-09-25T15:35:03Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T08:24:32.545342Z","submitted_at":"2019-09-25T15:35:03Z","title":"Reducing Transformer Depth on Demand with Structured Dropout"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 29 inbound Pith citation observations for arXiv:1909.11556."}