{"as_of":"2026-08-07T05:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cf74a693517e91f6e7765f0f38309bb3301345c3df67f2966278a1745147b63f","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T00:41:45.400805Z","state":"measured"},{"denominator":49,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":49,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.00129/citation-record","integrity":"/paper/2608.00129/integrity","json":"/paper/2608.00129/citation-record.json","paper":"/paper/2608.00129"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:41.845238Z","title":"Maximum likelihood estimation of intrinsic dimension,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:41.845238Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:ce65ba147ccfc34f8a8f4e64425b26eda58227f6c2f65d980731e7d222c5e06c","observation_id":"54d30d06-2e1a-4f50-8a32-23f85b8719b7","resolution":{"observed_at":"2026-08-04T00:41:41.845238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:41.915913Z","title":"Estimating the intrinsic dimension of datasets by a minimal neighborhood information,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:41.915913Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:e78bd65756475fcd9f421c0d0737fd6c007a1484f2ac08416864ba2f780fed91","observation_id":"29ece130-2877-44a6-b02b-1234ae9b39a4","resolution":{"observed_at":"2026-08-04T00:41:41.915913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:41.996537Z","title":"Intrinsic dimension of data representations in deep neural networks,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:41.996537Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:cc262dd4ea73fbfd33c49005ed5a234f3b4056bf7eb9ff0e61e6f48dce912d51","observation_id":"25f6940a-6c0e-428c-9e09-f32dd68950bb","resolution":{"observed_at":"2026-08-04T00:41:41.996537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.102457Z","title":"Rethinking feature-based knowledge distillation for face recognition,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.102457Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:e67127653c7cdcd069784739b83651013431b047db78567851f85b1d83d28c63","observation_id":"bdc76e4b-8641-430b-9016-aa33edab9801","resolution":{"observed_at":"2026-08-04T00:41:42.102457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.191100Z","title":"Understand- ing deep learning requires rethinking generalization,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.191100Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:b6f5ae8458a1adf3f7ecdf923867f542ebaae355efc19f6cab2e13855c14061b","observation_id":"48b77857-2091-4507-98c8-9129c4152d58","resolution":{"observed_at":"2026-08-04T00:41:42.191100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1511.06068","last_updated":"2016-06-10T10:59:37Z","snapshot_observed_at":"2026-07-06T04:37:02.350417Z","submitted_at":"2015-11-19T06:23:09Z","title":"Reducing Overfitting in Deep Networks by Decorrelating Representations","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1511.06068","snapshot_observed_at":"2026-08-04T00:41:42.271034Z","title":"Reduc- ing overfitting in deep networks by decorrelating representations,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.271034Z"},"links":{"cited_paper":"/paper/1511.06068","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:6af422b8f48a4e51d287a7d1293cd8b934468c7648879786055e93144ea36fc9","observation_id":"7dda9964-5885-400d-bff1-b60be9e45c53","resolution":{"observed_at":"2026-08-04T00:41:42.271034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.350556Z","title":"Rethinking the value of network pruning,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.350556Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:5b793a56bf094c5321016acbbdae695e1417d7e15070911d10dca627979b9f78","observation_id":"58c69924-75d9-4e4b-a4bc-515c942ec7e7","resolution":{"observed_at":"2026-08-04T00:41:42.350556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.409843Z","title":"Separability and geometry of object manifolds in deep neural networks,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.409843Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:92328e6faf47f2aca6a4258d04506c485b4d0a5373dc4e847fe65f8a581aa090","observation_id":"45e9a17d-7924-464e-a35c-f35a08211f40","resolution":{"observed_at":"2026-08-04T00:41:42.409843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.493973Z","title":"The cityscapes dataset for semantic urban scene understanding,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.493973Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:b856cee61b9500a532a2126f7f2caaad80680824a937f4ff0c8a24c64f535d67","observation_id":"a8733015-ea0e-4630-aec8-c0bfda921dc0","resolution":{"observed_at":"2026-08-04T00:41:42.493973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.572258Z","title":"Segnet: A deep con- volutional encoder-decoder architecture for image segmentation,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.572258Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:9c2ce5c509130b31a76a12f7f3526ef591ecea039a5d60b05e73941ee596727d","observation_id":"42877d43-5794-4e0e-adfb-24ae6d70e468","resolution":{"observed_at":"2026-08-04T00:41:42.572258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.631420Z","title":"Deep residual learning for image recognition,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.631420Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:56b91b64b9b4158d4d2e11ab7a832ac98e5a06edd249aeabb7e65130fc3b7dcb","observation_id":"cd7895d0-b1ce-4f2e-8de5-f244d387d43e","resolution":{"observed_at":"2026-08-04T00:41:42.631420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.685429Z","title":"Learning multiple layers of features from tiny images,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.685429Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:2a573b8467eb5d9636a70a878be9ed48fc74771e80d4516d4645e3782689011f","observation_id":"44bdb5af-71f4-4d98-a10b-e43596ca69e5","resolution":{"observed_at":"2026-08-04T00:41:42.685429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.740798Z","title":"Pruning filters for efficient convnets,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.740798Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:568519243760077d36352e5814df0c641a047dcf0a6c5e3458f149a4f7e4d561","observation_id":"0e816b68-970f-4911-a5b8-0d430d0b3cd0","resolution":{"observed_at":"2026-08-04T00:41:42.740798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.820937Z","title":"Distilling the knowledge in a neural network,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.820937Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:6c34c1b9fac48bc185a332aabec86a2e9a56b46d8488ae922088d2df1b07ddd7","observation_id":"0ffa45fa-ac20-445a-85b7-5254302d442a","resolution":{"observed_at":"2026-08-04T00:41:42.820937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.879307Z","title":"Improved knowledge distillation via teacher assis- tant,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.879307Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:afe0a9df77e363c509783770a37f923ac76085e6fe633bc452bd0bc8e0051a13","observation_id":"373c1a3d-c105-4419-af27-5063b675503e","resolution":{"observed_at":"2026-08-04T00:41:42.879307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:42.925512Z","title":"Knowledge distillation from a stronger teacher,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:42.925512Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:4c317c3ce849f4ba7a7231cd5c2f250131d170bfcacc0158e1a34be40dd4a904","observation_id":"c79ff685-4b04-4198-836a-b11fc996e34b","resolution":{"observed_at":"2026-08-04T00:41:42.925512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.036651Z","title":"Conflict-averse gradi- ent descent for multi-task learning,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.036651Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:1de4a06a9d633638b583a4f7c4c5de9c51c38970b7d4a9103759c3209b5346f3","observation_id":"3fd47dc6-395d-420e-ac95-b9205d43335e","resolution":{"observed_at":"2026-08-04T00:41:43.036651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.09355","last_updated":"2019-08-25T16:13:24Z","snapshot_observed_at":"2026-07-06T08:16:39.915426Z","submitted_at":"2019-08-25T16:13:24Z","title":"Patient Knowledge Distillation for BERT Model Compression","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.09355","snapshot_observed_at":"2026-08-04T00:41:43.093855Z","title":"Patient knowledge distillation for bert model compression,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.093855Z"},"links":{"cited_paper":"/paper/1908.09355","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:58f6f235b22ef6d396407a0aef18491934a389eba70828627ff6f569b4b289de","observation_id":"fd44600d-a514-43ac-ab1d-bc1b4840391a","resolution":{"observed_at":"2026-08-04T00:41:43.093855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.156779Z","title":"Curriculum temperature for knowledge distillation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.156779Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:7722a76fea997f092fecbc06d19b2cfe990930e6ffee9537a1b7333ead332fab","observation_id":"9030a84c-9a0b-4b18-bdc3-45a0dc5d431f","resolution":{"observed_at":"2026-08-04T00:41:43.156779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.227117Z","title":"Rethinking soft labels for knowledge distillation: A bias–variance tradeoff perspective,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.227117Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:f2937f30b41deb25861dcf2e6f5b3cb8f7abc6e7c85ac382b3689918c86c4f88","observation_id":"77ddb675-da69-4a63-9d3d-618c68925add","resolution":{"observed_at":"2026-08-04T00:41:43.227117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.305427Z","title":"Controllable dynamic multi- task architectures,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.305427Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:0d804434b78cd7ba2b29d287ca512b7f05cb8eca7c140f8a2be1fc20c0684f0b","observation_id":"b55e39f0-b369-4498-b6de-9205450a91e8","resolution":{"observed_at":"2026-08-04T00:41:43.305427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-04T00:41:43.369475Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.369475Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:de35e9f9817c3fdad8e2c883448b02a33ed3bf443af1b3c2d1e713f124a04c3a","observation_id":"8e0e6295-45e0-44b0-93c2-fb9a94c6e9a2","resolution":{"observed_at":"2026-08-04T00:41:43.369475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-04T00:41:43.429602Z","title":"Deepseek-v3 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.429602Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:a46af487b37a27ffe59f9271ee1d03e25e363dfa64c97aac2043aef5b6808ca7","observation_id":"1e1bafa9-c788-4ad8-a806-f43d9475171a","resolution":{"observed_at":"2026-08-04T00:41:43.429602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.469735Z","title":"Student customized knowledge distillation: Bridging the gap between student and teacher,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.469735Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:550b619c57f7c96296306435d71740240f69b300d666d0d27b9b9db6a5b9f3e6","observation_id":"7802ae2d-e4a9-49ff-947a-ac1424018c3f","resolution":{"observed_at":"2026-08-04T00:41:43.469735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.546896Z","title":"Gra- dient surgery for multi-task learning,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.546896Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:425f84a6c58c400b81de94f487b7774b7e3b57e843a6e00cbc849da4026c2114","observation_id":"0e4e2623-3070-4c1e-8f09-d624f806328d","resolution":{"observed_at":"2026-08-04T00:41:43.546896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.625895Z","title":"Famo: Fast adaptive multitask optimization,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.625895Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:952aab9fcfc2207cbffadb0302c52aff617c24642a084556de4931b833c81286","observation_id":"b2b3ec1b-8d84-4a5b-b008-30184620d1dc","resolution":{"observed_at":"2026-08-04T00:41:43.625895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.713723Z","title":"Knowledge distillation for multi-task learning,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.713723Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:d42c24a3aade47d31d34f0c381f38efee660062a7d7a01869d72d6849f54eac6","observation_id":"f0c87c78-f12d-4320-8373-6aa39e099c55","resolution":{"observed_at":"2026-08-04T00:41:43.713723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.791007Z","title":"Contrastive representation distilla- tion,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.791007Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:f49b6b0805b15f238fe3463e2e4ba42b3f05788757e44d29ec6fe921e6963c68","observation_id":"b4fd4dcb-10f9-4bb4-ac47-196bacfc23a7","resolution":{"observed_at":"2026-08-04T00:41:43.791007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.876346Z","title":"Cross-layer distillation with semantic calibration,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.876346Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:f1cf2b6bff187519d0a40ed05e03b3934855d929e9325e9af6eaf012100a5c90","observation_id":"c6c8d4a5-e74f-42c5-b6be-319e2b6f79a2","resolution":{"observed_at":"2026-08-04T00:41:43.876346Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:43.954965Z","title":"On the efficacy of knowledge distillation,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:43.954965Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:090dd639b9172e885be52fbb1f68991696b86939266f57007372bb7c1260f791","observation_id":"c966cd39-c4df-4366-b0ac-937dd634ee5f","resolution":{"observed_at":"2026-08-04T00:41:43.954965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.033714Z","title":"Segformer: Simple and efficient design for semantic segmentation with transformers,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.033714Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:f19ad063c787a1995ac9f5100ec2a7b2d5dd57f0e67ba302b734cb67db004735","observation_id":"ca0ad48d-4f1a-4f6e-87fc-6e37d5bb778c","resolution":{"observed_at":"2026-08-04T00:41:44.033714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-04T00:41:44.116618Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.116618Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:9a40f5866c5a16d5ae3ad3b97cf11dc93cee88d2a1dda84dd4d00399e5206a84","observation_id":"42ca9a07-8e5e-4dfd-9024-692f4cf1110f","resolution":{"observed_at":"2026-08-04T00:41:44.116618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.225038Z","title":"Distilbert, a distilled version of bert: smaller, faster, cheaper and lighter","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.225038Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:1e6cbbb5bb79e34b3e3c5bdb2ead49f2218b385f970191d874e8dced70361a44","observation_id":"66964b90-1d97-41b3-8022-b9ebe4035c7d","resolution":{"observed_at":"2026-08-04T00:41:44.225038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.301067Z","title":"Tinybert: Distilling bert for natural language understanding,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.301067Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:8a06fe198dea3dc9cbfa1dff175f3651024a6559a17254cd9a8a9092e997c808","observation_id":"a392c23c-ce01-4817-bcbd-cb71c25b26cb","resolution":{"observed_at":"2026-08-04T00:41:44.301067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.354192Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.354192Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:95ae6bedb5d199c06d5ab0d39fd6e7bd6b5ef00b05a8814088edc0248d87d9b0","observation_id":"f03b68cc-c611-45e2-ba1c-45702754f605","resolution":{"observed_at":"2026-08-04T00:41:44.354192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.438195Z","title":"Learning multiple dense prediction tasks from partially annotated data,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.438195Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:4b9bbc02f5e9c5095c6b9bd733d8e6bd53154d9a4a22d5b009d9298c12272be3","observation_id":"6fef3917-e48c-48b8-9321-6fe7bd0752a2","resolution":{"observed_at":"2026-08-04T00:41:44.438195Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.519861Z","title":"Does knowledge distillation really work?","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.519861Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:627ea0b7a1f878e2bde65eb75f4ee1163de42b287785743294b09e638a825001","observation_id":"67fea1e5-6302-4981-91f4-57d8090260ad","resolution":{"observed_at":"2026-08-04T00:41:44.519861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.600707Z","title":"The platonic representation hypothesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.600707Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:7b183733aa10539183053534d5ab44ffabe24223d854a6d36d673f22c6e87c30","observation_id":"ca4ff926-6eff-49ed-824b-6aa7d5c57072","resolution":{"observed_at":"2026-08-04T00:41:44.600707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.686417Z","title":"Densely guided knowledge distillation using multiple teacher assistants,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.686417Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:40b2f8d164df9ec3a67eca7a02377b0406123ae6a98052a145a72a3c9004f9f2","observation_id":"235a84e6-e2e1-4861-8819-7fc7b363a5d4","resolution":{"observed_at":"2026-08-04T00:41:44.686417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.771420Z","title":"Monotakd: Teaching assistant knowledge distillation for monocular 3d object detection,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.771420Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:842bf212a7d91da4d68b0719c1790f745851a104d4a1f0da2050147edda86c16","observation_id":"62b3711e-258e-431c-90dd-e675931152a3","resolution":{"observed_at":"2026-08-04T00:41:44.771420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.849892Z","title":"An overview of statistical learning theory,","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.849892Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:99475718f9c4c9615baf52c29cc1d40f120bfa13f8d032ac7b587b0c1858df42","observation_id":"245f2a91-cea5-4728-829c-5f9810986897","resolution":{"observed_at":"2026-08-04T00:41:44.849892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1511.03643","last_updated":"2016-02-26T02:21:52Z","snapshot_observed_at":"2026-07-06T04:36:10.443201Z","submitted_at":"2015-11-11T20:27:54Z","title":"Unifying distillation and privileged information","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1511.03643","snapshot_observed_at":"2026-08-04T00:41:44.932796Z","title":"Unifying dis- tillation and privileged information,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.932796Z"},"links":{"cited_paper":"/paper/1511.03643","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:9887247a4c35f51f21dd9bd437c38cfccc78ba5d3cdef5544fbec07eaaa1a3fd","observation_id":"b3e74537-92e8-4095-9887-e29665bada42","resolution":{"observed_at":"2026-08-04T00:41:44.932796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:44.996213Z","title":"On the difficulty of training recurrent neural networks,","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:44.996213Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:fa657c876daab0618268c26bbb65a248a2b772079dbcc5b4410a5a8d81e6df88","observation_id":"51c7dd8b-cb5a-4686-abd5-5a7c45066fa5","resolution":{"observed_at":"2026-08-04T00:41:44.996213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1812.06162","last_updated":"2018-12-14T20:49:09Z","snapshot_observed_at":"2026-07-06T07:21:19.842962Z","submitted_at":"2018-12-14T20:49:09Z","title":"An Empirical Model of Large-Batch Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1812.06162","snapshot_observed_at":"2026-08-04T00:41:45.069859Z","title":"An empirical model of large-batch training,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:45.069859Z"},"links":{"cited_paper":"/paper/1812.06162","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:f40859a7de42a3b29307fc8580adc4d64480df7a987de9343fc8cd1cbb283c6d","observation_id":"ff3e5545-6769-478e-9994-bc2755430aaa","resolution":{"observed_at":"2026-08-04T00:41:45.069859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:45.125272Z","title":"Gradnorm: Gradient normalization for adaptive loss balancing in deep multitask networks,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:45.125272Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:b83e75fff4bbdc535bc0a8c8a92b5e40da1994523ede117bc9c88284e03b3d74","observation_id":"f5941856-39a2-4413-b3b9-d8d2c1d858bc","resolution":{"observed_at":"2026-08-04T00:41:45.125272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.04532","last_updated":"2020-07-09T03:23:10Z","snapshot_observed_at":"2026-07-06T09:36:43.462530Z","submitted_at":"2020-07-09T03:23:10Z","title":"A Study of Gradient Variance in Deep Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.04532","snapshot_observed_at":"2026-08-04T00:41:45.205816Z","title":"A study of gradient variance in deep learning,","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:45.205816Z"},"links":{"cited_paper":"/paper/2007.04532","citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:2e5c73150673377c269162bf7e4362fd5b64585529fbcbe0122b5bbe8ed9b853","observation_id":"707b3328-6f5e-4522-bb94-1ef4f70af353","resolution":{"observed_at":"2026-08-04T00:41:45.205816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:45.282613Z","title":"Decoupled knowledge distillation,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:45.282613Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:41843dd07539c7eead9c9e03b75fdf0934cf1ad76deef2703e21e3610bedd432","observation_id":"123b12bd-af4b-4ec0-b70c-5fe2a9797b6e","resolution":{"observed_at":"2026-08-04T00:41:45.282613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:45.341638Z","title":"End-to-end multi-task learning with attention,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:45.341638Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:bb70c1d7444c1bfb980791d1af1c23bee18621bdef5fa903201da0f1382fc0d8","observation_id":"3836c3af-2be0-4d7d-8bfb-615999f8f3aa","resolution":{"observed_at":"2026-08-04T00:41:45.341638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T00:41:45.400805Z","title":"Inverted pyramid multi-task transformer for dense scene understanding,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-04T00:41:45.400805Z"},"links":{"citing_paper":"/paper/2608.00129"},"observation_digest":"sha256:af1cf96d16f05f74c6fbc4cc36db93a14619c10a4abc840fcd7736b5229e2ff8","observation_id":"011fa260-d639-435b-8e6f-7e1305d9a681","resolution":{"observed_at":"2026-08-04T00:41:45.400805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.00129","last_updated":"2026-07-31T13:19:08Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T23:11:55.135752Z","submitted_at":"2026-07-31T13:19:08Z","title":"Progressive$^2$: A Teacher-Student Progressive Co-Evolving Knowledge Distillation Method for Substantial Model Compression"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":49,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 0 inbound Pith citation observations for arXiv:2608.00129."}