{"as_of":"2026-08-08T21:14:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f4d3261c0ca3a07465952a0e6074ade4499444ffaa291e911355e8368a5f4033","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":39,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T14:29:13.847195Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":5,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2412.14590","last_updated":"2026-04-22T15:43:48Z","snapshot_observed_at":"2026-08-03T01:50:37.847212Z","submitted_at":"2024-12-19T07:15:15Z","title":"MixLLM: LLM Quantization with Global Mixed-precision between Output-features and Highly-efficient System Design","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T06:56:51.829741Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2412.14590"},"observation_digest":"sha256:2bb3acac3f0289f36cc75db1a0b685f0c354485ffbb021a0459c3b8a81623d9f","observation_id":"15924654-a6e9-4bb0-a154-96407715c746","resolution":{"observed_at":"2026-05-23T06:57:40.285477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-08T14:29:13.847195Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06766","last_updated":"2025-02-12T15:55:37Z","snapshot_observed_at":"2026-08-08T14:22:26.963746Z","submitted_at":"2025-02-10T18:47:04Z","title":"Exploiting Sparsity for Long Context Inference: Million Token Contexts on Commodity GPUs","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-08T14:29:13.847195Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2502.06766"},"observation_digest":"sha256:df1bb26c3714d329fba51b73472872d26d02b3169d5e5285d5df87d535f77f78","observation_id":"8278ec27-444a-4a53-b670-5bf17e096aa5","resolution":{"observed_at":"2026-08-08T14:29:13.847195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-08T11:39:09.477311Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.07780","last_updated":"2026-07-15T12:46:32Z","snapshot_observed_at":"2026-08-08T11:32:49.912375Z","submitted_at":"2025-02-11T18:59:35Z","title":"DarwinLM: Evolutionary Structured Pruning of Large Language Models","version":4},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-08T11:39:09.477311Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2502.07780"},"observation_digest":"sha256:dc9ceabe5ca81e581a29888445651e921e1d2c684c9462af321e447923a86152","observation_id":"d0879de0-20b5-4512-9560-2606bf0d14a3","resolution":{"observed_at":"2026-08-08T11:39:09.477311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-08T13:44:00.495709Z","title":"The unreasonable ineffectiveness of the deeper layers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07832","last_updated":"2025-02-11T00:21:40Z","snapshot_observed_at":"2026-08-08T13:37:47.274283Z","submitted_at":"2025-02-11T00:21:40Z","title":"SHARP: Accelerating Language Model Inference by SHaring Adjacent layers with Recovery Parameters","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T13:44:00.495709Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2502.07832"},"observation_digest":"sha256:1f6fc1aca4f6bb13af187791dfdfdf3d729f70a9ce01261b3cadc69d78c6095e","observation_id":"a2a0077f-9906-4d40-a7c2-332cf89e403b","resolution":{"observed_at":"2026-08-08T13:44:00.495709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-07T22:34:08.022953Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.12170","last_updated":"2025-05-28T12:57:47Z","snapshot_observed_at":"2026-08-07T23:11:12.819339Z","submitted_at":"2025-02-13T10:26:27Z","title":"MUDDFormer: Breaking Residual Bottlenecks in Transformers via Multiway Dynamic Dense Connections","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T22:34:08.022953Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2502.12170"},"observation_digest":"sha256:c4cb769cf465e7364714ca58e822f509f4197f74991d72343ad7665537d23455","observation_id":"aed77fdf-3f4e-48ae-b846-e454c23d7916","resolution":{"observed_at":"2026-08-07T22:34:08.022953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-07T15:37:00.813610Z","title":"Gromov, K","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14467","last_updated":"2025-05-20T15:01:56Z","snapshot_observed_at":"2026-08-07T23:37:44.957755Z","submitted_at":"2025-05-20T15:01:56Z","title":"Void in Language Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T15:37:00.813610Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2505.14467"},"observation_digest":"sha256:a6ec92d32a364613972a8ea64b7a69c839d96361dbd6f73c664b173644fe8e08","observation_id":"16d7c48b-a93a-4468-b8f5-be1252e80b84","resolution":{"observed_at":"2026-08-07T15:37:00.813610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-07T14:48:02.216637Z","title":"The unreasonable ineffectiveness of the deeper layers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17626","last_updated":"2025-05-23T08:36:56Z","snapshot_observed_at":"2026-08-07T23:10:32.302983Z","submitted_at":"2025-05-23T08:36:56Z","title":"Leveraging Stochastic Depth Training for Adaptive Inference","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:48:02.216637Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2505.17626"},"observation_digest":"sha256:2e88504473c16a16895aa79830983eed3d4b9fad818cd36a9a0a513e5a29c4b0","observation_id":"65109521-ad22-4ee9-9643-971dc58d939b","resolution":{"observed_at":"2026-08-07T14:48:02.216637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-07T10:51:51.491364Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04179","last_updated":"2025-06-04T17:26:31Z","snapshot_observed_at":"2026-08-08T19:35:37.122227Z","submitted_at":"2025-06-04T17:26:31Z","title":"SkipGPT: Dynamic Layer Pruning Reinvented with Token Awareness and Module Decoupling","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T10:51:51.491364Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2506.04179"},"observation_digest":"sha256:70a5e0b5982c869eac2bd55765611928c4d1368bd956be250d51b9afb3444023","observation_id":"c4e4d684-fa8f-416f-b29c-851b1e4d532f","resolution":{"observed_at":"2026-08-07T10:51:51.491364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T22:53:11.013316Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20480","last_updated":"2025-06-25T14:24:59Z","snapshot_observed_at":"2026-08-08T05:07:43.279449Z","submitted_at":"2025-06-25T14:24:59Z","title":"GPTailor: Large Language Model Pruning Through Layer Cutting and Stitching","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T22:53:11.013316Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2506.20480"},"observation_digest":"sha256:881c9274afa5b0c2e01d8326381e69ce0c073de42b50b7a6b9f3855bfdee2873","observation_id":"b2b52689-eac5-4880-893e-9e4446e96386","resolution":{"observed_at":"2026-08-06T22:53:11.013316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T22:21:09.909236Z","title":"The unreasonable ineffectiveness of the deeper layers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22049","last_updated":"2025-07-03T16:54:09Z","snapshot_observed_at":"2026-08-07T23:11:38.262545Z","submitted_at":"2025-06-27T09:45:15Z","title":"GPAS: Accelerating Convergence of LLM Pretraining via Gradient-Preserving Activation Scaling","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:21:09.909236Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2506.22049"},"observation_digest":"sha256:6c0870df7ab0ecf094185941cd4b99a4ea2594fc793f6e264233b48c33229375","observation_id":"b716bc9a-db6d-4c36-bcd7-c917ef91e4cc","resolution":{"observed_at":"2026-08-06T22:21:09.909236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T22:14:19.114883Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.22389","last_updated":"2025-06-27T16:57:59Z","snapshot_observed_at":"2026-08-08T09:56:16.215858Z","submitted_at":"2025-06-27T16:57:59Z","title":"Towards Distributed Neural Architectures","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T22:14:19.114883Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2506.22389"},"observation_digest":"sha256:ad5e04bfc252a44819e9cd224c5c171a1340ebe5d3dfc3f96ffbb7c747fdc6e3","observation_id":"a0a084e5-e6ef-4663-85d2-3565f95c0be8","resolution":{"observed_at":"2026-08-06T22:14:19.114883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T19:14:25.884327Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06203","last_updated":"2025-07-10T16:43:36Z","snapshot_observed_at":"2026-08-07T04:57:37.201438Z","submitted_at":"2025-07-08T17:29:07Z","title":"A Survey on Latent Reasoning","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T19:14:25.884327Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2507.06203"},"observation_digest":"sha256:2786024807002bdb4356bc4afd970fcb543f56526d91b4b000791afd88864c13","observation_id":"c39e6cb2-95ed-417a-99fd-7ee09908d0f3","resolution":{"observed_at":"2026-08-06T19:14:25.884327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T18:39:00.502583Z","title":"org/abs/2403.17887 ([n","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08064","last_updated":"2026-06-06T18:51:46Z","snapshot_observed_at":"2026-08-06T18:26:57.432317Z","submitted_at":"2025-07-10T16:47:25Z","title":"PUMA: Layer-Pruned Language Model for Efficient Unified Multimodal Retrieval with Modality-Adaptive Learning","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T18:39:00.502583Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2507.08064"},"observation_digest":"sha256:a650f60ce6eb5f39e10a0876c42d29c06e108606dd07305d2f948222a235e5cd","observation_id":"4779f547-3305-4fa8-922a-a5d87963794a","resolution":{"observed_at":"2026-08-06T18:39:00.502583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T10:55:18.132568Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.23362","last_updated":"2025-07-31T09:17:53Z","snapshot_observed_at":"2026-08-08T14:50:30.917971Z","submitted_at":"2025-07-31T09:17:53Z","title":"Short-LVLM: Compressing and Accelerating Large Vision-Language Models by Pruning Redundant Layers","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T10:55:18.132568Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2507.23362"},"observation_digest":"sha256:02d925229bf05c50c132b2f0b0c61a4beefe54cf914fefc60d0cd973d814159b","observation_id":"bcab9834-a370-4cff-9f4f-0d1f01636d0d","resolution":{"observed_at":"2026-08-06T10:55:18.132568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-06T05:11:11.524583Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.02128","last_updated":"2025-08-04T07:22:36Z","snapshot_observed_at":"2026-08-08T06:49:38.643278Z","submitted_at":"2025-08-04T07:22:36Z","title":"Amber Pruner: Leveraging N:M Activation Sparsity for Efficient Prefill in Large Language Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T05:11:11.524583Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2508.02128"},"observation_digest":"sha256:e1a772c907039da2f49b2b76c81d678f3281a3a4fc8fa2888121e8c9930f1e1a","observation_id":"428b471c-e54d-40c7-bb57-69bac7a49e01","resolution":{"observed_at":"2026-08-06T05:11:11.524583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2602.01997","last_updated":"2026-08-06T07:02:58Z","snapshot_observed_at":"2026-08-08T20:15:35.368248Z","submitted_at":"2026-02-02T11:57:22Z","title":"On the Limits of Layer Pruning for Generative Reasoning in Large Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T08:40:24.822863Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2602.01997"},"observation_digest":"sha256:3f004003076c33b922997e3554d8bd2318d853ad623764fca3663561086c0eb8","observation_id":"df68b59e-b75e-416b-9d0d-c110f7247e5c","resolution":{"observed_at":"2026-05-16T08:40:46.167908Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-03T04:07:45.518722Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.05970","last_updated":"2026-05-31T19:48:07Z","snapshot_observed_at":"2026-08-07T15:15:21.459993Z","submitted_at":"2026-02-05T18:22:41Z","title":"Inverse Depth Scaling From Most Layers Being Similar","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T04:07:45.518722Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2602.05970"},"observation_digest":"sha256:e93ff197cbecee7fce5e0459ac73f1c4189a8059a369331029b14e501111b38d","observation_id":"e0df9383-f5c5-4323-bde1-0dd286d3095f","resolution":{"observed_at":"2026-08-03T04:07:45.518722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2603.15031","last_updated":"2026-03-16T09:32:21Z","snapshot_observed_at":"2026-08-02T08:46:00.749789Z","submitted_at":"2026-03-16T09:32:21Z","title":"Attention Residuals","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T06:39:04.312270Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2603.15031"},"observation_digest":"sha256:0f80cf4115563d8699c588d7559f461230584f3b8dac8d780b14f98f14ee50f5","observation_id":"ff7b1b89-f86b-40df-af12-51f254be6010","resolution":{"observed_at":"2026-05-21T06:39:04.388399Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-07-14T20:29:33.439034Z","title":"The unreason- able ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.15389","last_updated":"2026-06-27T04:53:36Z","snapshot_observed_at":"2026-08-01T17:36:36.240938Z","submitted_at":"2026-03-16T15:04:16Z","title":"When Does Sparsity Mitigate the Curse of Depth in LLMs","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-14T20:29:33.439034Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2603.15389"},"observation_digest":"sha256:e6cf1cd8a5d24aac0cfc114e0898cb7e24fc1acf316909a672787f299f8048d5","observation_id":"05093803-5c9d-4837-b839-8b3b9182506b","resolution":{"observed_at":"2026-07-14T20:29:33.439034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-07-06T23:00:37.144068Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:c1df4c555c177dd3685f9bee072a49e0993e20a8ee8f8d8075da6f09f4eb4a46","observation_id":"d33c3ed4-d16b-4b5c-9096-fdb75cea654a","resolution":{"observed_at":"2026-05-11T10:16:07.468222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2604.17224","last_updated":"2026-04-19T03:11:30Z","snapshot_observed_at":"2026-08-03T11:22:11.323337Z","submitted_at":"2026-04-19T03:11:30Z","title":"LASER: Low-Rank Activation SVD for Efficient Recursion","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T07:12:34.363456Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2604.17224"},"observation_digest":"sha256:0da47f38e39220afa3c5224c334f46dfa0546acb8e358c455495dc5f96109b19","observation_id":"e8104843-579d-42a9-9d15-eb0b8db47ea3","resolution":{"observed_at":"2026-05-10T09:23:37.409033Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2604.20682","last_updated":"2026-04-22T15:31:46Z","snapshot_observed_at":"2026-07-06T23:07:14.243878Z","submitted_at":"2026-04-22T15:31:46Z","title":"Variance Is Not Importance: Structural Analysis of Transformer Compressibility Across Model Scales","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T01:44:42.989053Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2604.20682"},"observation_digest":"sha256:e5abda688b21dfcce7536b7ae2974f9834d0f373c4294c84748d4e3b2b8c7420","observation_id":"36982239-45d0-44d2-8e0a-541c046845e1","resolution":{"observed_at":"2026-05-11T13:26:04.417968Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2605.04971","last_updated":"2026-05-06T14:27:18Z","snapshot_observed_at":"2026-07-06T23:17:38.226365Z","submitted_at":"2026-05-06T14:27:18Z","title":"Why Geometric Continuity Emerges in Deep Neural Networks: Residual Connections and Rotational Symmetry Breaking","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-08T17:21:23.468992Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2605.04971"},"observation_digest":"sha256:6215c648448dba6fcae79bea50c09848b01b9b7eb51fb37f5713a89318950064","observation_id":"da4b0a3c-dd7c-472f-8b1e-611296fcf2c8","resolution":{"observed_at":"2026-05-11T17:41:06.430899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2605.07271","last_updated":"2026-05-08T05:35:32Z","snapshot_observed_at":"2026-07-06T23:19:41.053744Z","submitted_at":"2026-05-08T05:35:32Z","title":"Understanding Performance Collapse in Layer-Pruned Large Language Models via Decision Representation Transitions","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-11T02:23:52.589354Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2605.07271"},"observation_digest":"sha256:b782feb3d54d74c573aef14f9314c805c45d3fed03ef1d8d9ad6e9f5b5eebe28","observation_id":"968f2024-b415-4611-830c-e3e2d21054b6","resolution":{"observed_at":"2026-05-11T02:25:53.883150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2605.25344","last_updated":"2026-08-02T13:40:15Z","snapshot_observed_at":"2026-08-06T23:24:41.987610Z","submitted_at":"2026-05-25T02:00:41Z","title":"A Hamiltonian-Inspired Local-Operator Ansatz for Slimming Large Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-29T23:00:21.397982Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2605.25344"},"observation_digest":"sha256:527754470e0166ae363a864902bc765853382f1322c7caf624e610fab02acb31","observation_id":"9edb290a-8f87-4e97-beaf-00ab32135e20","resolution":{"observed_at":"2026-06-29T23:04:01.068662Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-04T05:01:02.976226Z","title":"& Roberts, D","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.25344","last_updated":"2026-08-02T13:40:15Z","snapshot_observed_at":"2026-08-06T23:24:41.987610Z","submitted_at":"2026-05-25T02:00:41Z","title":"A Hamiltonian-Inspired Local-Operator Ansatz for Slimming Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T05:01:02.976226Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2605.25344"},"observation_digest":"sha256:a46c4951d10bf7d48cefc2a3e5ad6b3a2c19e1517c4a4afec8760c7cac96afdf","observation_id":"0f089641-195d-4c4e-863e-279db3fc46d7","resolution":{"observed_at":"2026-08-04T05:01:02.976226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2605.26496","last_updated":"2026-05-26T03:19:04Z","snapshot_observed_at":"2026-07-06T23:36:20.072253Z","submitted_at":"2026-05-26T03:19:04Z","title":"Dense2MoE: Pushing the Pareto Frontier of On-Device LLMs via Unified Pruning and Upcycling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T19:34:17.270161Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2605.26496"},"observation_digest":"sha256:307ae766f3075829e0d08065fd9ea400d4b3dd04c6fe1db30651bb079c9117e8","observation_id":"6b9f4ddb-81d9-427a-9268-c1049d67dde0","resolution":{"observed_at":"2026-06-29T19:43:55.034985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2606.19150","last_updated":"2026-06-17T14:56:27Z","snapshot_observed_at":"2026-08-06T08:43:49.965943Z","submitted_at":"2026-06-17T14:56:27Z","title":"Complementary Attention Head Pruning for Efficient Transformers","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-26T20:56:52.981010Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.19150"},"observation_digest":"sha256:120c755d1089c8ba2838107f10ae07d7bcd4e4ae14af28785a83b2225b85f13b","observation_id":"2c2190d2-c753-4317-80b6-fb56bc47b5c2","resolution":{"observed_at":"2026-07-04T00:49:18.756029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2606.20246","last_updated":"2026-06-18T13:57:12Z","snapshot_observed_at":"2026-08-05T09:17:49.575980Z","submitted_at":"2026-06-18T13:57:12Z","title":"Finetuning Vision-Language-Action Models Requires Fewer Layers Than You Think","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T16:59:28.243515Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.20246"},"observation_digest":"sha256:2ccb47eddac373723ccc5728ca6cd5b620f35ff5319f5b900a83fb229fb47d70","observation_id":"906b38bd-28c8-4533-8690-63b81b2d6f74","resolution":{"observed_at":"2026-07-04T04:29:34.938693Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2606.23670","last_updated":"2026-06-22T17:56:25Z","snapshot_observed_at":"2026-08-08T15:39:37.831663Z","submitted_at":"2026-06-22T17:56:25Z","title":"Tapered Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-26T09:11:20.341634Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.23670"},"observation_digest":"sha256:c0e5aa8029d64757bed610dec87aec75a40d647430610aa35671c68009f661a7","observation_id":"038e7b6e-7223-4dda-bee5-2689f8499486","resolution":{"observed_at":"2026-07-04T10:09:43.999154Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2606.25008","last_updated":"2026-06-23T17:46:30Z","snapshot_observed_at":"2026-07-31T03:01:22.674756Z","submitted_at":"2026-06-23T17:46:30Z","title":"Neural Scaling Universality: If Exponents Are Fixed, Time to Understand Coefficients","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-25T23:45:54.283436Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.25008"},"observation_digest":"sha256:6f7e6362ada798f8bc24e00eb9787fdd8221c2e506749b841938b21923d5ae57","observation_id":"bb2b8ee1-e747-4e0f-9231-b0b9414c71b2","resolution":{"observed_at":"2026-07-04T17:20:00.869307Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2606.26538","last_updated":"2026-06-25T02:25:00Z","snapshot_observed_at":"2026-07-07T00:00:51.779266Z","submitted_at":"2026-06-25T02:25:00Z","title":"CascadeFormer: Depth-Tapered Transformers Motivated by Gradient Fan-in Asymmetry","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-26T05:22:26.818078Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.26538"},"observation_digest":"sha256:ef95e8bf24b639950966bfcfdf3dd4090e0d0699b87862a5697382f6bf92b0ac","observation_id":"75daf596-2a9d-485b-bb1d-3b0c5e5ed129","resolution":{"observed_at":"2026-06-26T05:29:00.076990Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2606.30813","last_updated":"2026-06-29T18:37:34Z","snapshot_observed_at":"2026-08-04T02:17:52.840024Z","submitted_at":"2026-06-29T18:37:34Z","title":"Gradient Smoothing: Coupling Layer-wise Updates for Improved Optimization","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-01T06:36:48.524846Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.30813"},"observation_digest":"sha256:c4159644c452ee53e397f18259c844aa684d5d4b6b873051fff8c19c051e5044","observation_id":"089336c5-735d-4e29-a588-97efa8dd67c0","resolution":{"observed_at":"2026-07-01T09:25:41.186898Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2606.31796","last_updated":"2026-07-23T16:29:18Z","snapshot_observed_at":"2026-08-03T13:02:28.914044Z","submitted_at":"2026-06-30T15:14:38Z","title":"CHERRY: Compressed Hierarchical Experts with Recurrent Representational Yield","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-01T05:46:40.510955Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.31796"},"observation_digest":"sha256:c1aa4c34018ac6a8af0335c04445c4ccc779c61863d5cc5115587fa0f74452a7","observation_id":"afafa4bd-10a8-460a-8c10-0b46a478daa9","resolution":{"observed_at":"2026-07-01T10:15:44.054375Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-02T09:24:12.859247Z","title":"& Roberts, D","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.31796","last_updated":"2026-07-23T16:29:18Z","snapshot_observed_at":"2026-08-03T13:02:28.914044Z","submitted_at":"2026-06-30T15:14:38Z","title":"CHERRY: Compressed Hierarchical Experts with Recurrent Representational Yield","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T09:24:12.859247Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2606.31796"},"observation_digest":"sha256:7352f7405814abdda896d980293ca0a37809bc36fafd65b6f9bdf109e3b79c8e","observation_id":"1b3ff6bd-0d27-4ead-bada-7607901cac3b","resolution":{"observed_at":"2026-08-02T09:24:12.859247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-07-11T23:58:47.097757Z","title":"arXiv preprint arXiv:2403.17887 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03784","last_updated":"2026-07-04T09:17:25Z","snapshot_observed_at":"2026-08-06T21:45:38.793670Z","submitted_at":"2026-07-04T09:17:25Z","title":"Rethinking Depth Pruning for Vision Transformers: A Heterogeneity-Aware Perspective","version":1},"reference_index":128,"source":"arxiv_source","source_observed_at":"2026-07-11T23:58:47.097757Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2607.03784"},"observation_digest":"sha256:b35d6e06ea3a02ec1908b338afb110fde86e3f4bdb0c4c21d2f7f2b4f78a61f9","observation_id":"448ee395-7f79-4a6c-a86d-f39707d2bc37","resolution":{"observed_at":"2026-07-11T23:58:47.097757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-07-11T15:58:40.406682Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv:2403.17887, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04640","last_updated":"2026-07-06T03:51:24Z","snapshot_observed_at":"2026-08-06T04:59:45.512540Z","submitted_at":"2026-07-06T03:51:24Z","title":"Wrong Before Right: Late Rescue and Interface Failure in Aligned Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-11T15:58:40.406682Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2607.04640"},"observation_digest":"sha256:143af838bc7729737512f3276467a23617c37c83b9578b914ce14eb224331d3e","observation_id":"db6df2d3-4bbb-45c9-9a50-b67f2d74f697","resolution":{"observed_at":"2026-07-11T15:58:40.406682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-02T14:51:05.502290Z","title":"Gromov, K","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14103","last_updated":"2026-05-06T13:12:00Z","snapshot_observed_at":"2026-08-07T23:12:12.290593Z","submitted_at":"2026-05-06T13:12:00Z","title":"Latent Communication Between Language Model Agents: Channels, Alignment, and the Limits of Text","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T14:51:05.502290Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2607.14103"},"observation_digest":"sha256:456bdd8918401ceb81822b32529b48a092ead2887ef6c8ee27d3b6b1d6d7df12","observation_id":"7a05f125-ea59-4a1e-9499-0f213ae44cec","resolution":{"observed_at":"2026-08-02T14:51:05.502290Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-01T03:15:58.833724Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25180","last_updated":"2026-07-28T01:13:37Z","snapshot_observed_at":"2026-08-06T15:30:04.436114Z","submitted_at":"2026-07-28T01:13:37Z","title":"Bekko Embedding: Parameter-Efficient Multilingual Retrieval with Ultra-Compact Encoders","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-01T03:15:58.833724Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2607.25180"},"observation_digest":"sha256:af293911cafee80ee0b48e756489c564ed6da872c421ad9cdc32a9ae3a6b2a33","observation_id":"d2735368-4d99-448e-977c-6c8e1e2645b6","resolution":{"observed_at":"2026-08-01T03:15:58.833724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.17887/citation-record","integrity":"/paper/2403.17887/integrity","json":"/paper/2403.17887/citation-record.json","paper":"/paper/2403.17887"},"outbound":[],"paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-03T06:33:46.668360Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 39 inbound Pith citation observations for arXiv:2403.17887."}