{"as_of":"2026-08-06T13:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:85ca5f39688bf834041087bf0a2a2072f4703f95af870ff02822d69d797dd712","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T20:45:00.558941Z","state":"measured"},{"denominator":66,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":66,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.08376/citation-record","integrity":"/paper/2509.08376/integrity","json":"/paper/2509.08376/citation-record.json","paper":"/paper/2509.08376"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.235614Z","title":"Emergence of in- variance and disentanglement in deep representations.The Journal of Machine Learning Research, 19(1):1947–1980,","venue":null,"work_id":"a28403c6-7c62-4425-b43e-b3588ac944a1","year":1947},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T20:44:59.710879Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:a9380ca6a145834819edf88cadb0a6f2631c942d29c1af30ac1261c75ce70682","observation_id":"b3d2fe4b-137e-4cad-b2e4-86a519fa0b96","resolution":{"observed_at":"2026-08-04T20:45:01.238598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.00496","last_updated":"2018-10-28T14:29:46Z","snapshot_observed_at":"2026-07-06T06:58:51.877372Z","submitted_at":"2018-09-03T08:38:34Z","title":"LRS3-TED: a large-scale dataset for visual speech recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.00496","snapshot_observed_at":"2026-08-04T20:44:59.780638Z","title":"Lrs3-ted: a large-scale dataset for visual speech recog- nition.arXiv preprint arXiv:1809.00496, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T20:44:59.780638Z"},"links":{"cited_paper":"/paper/1809.00496","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:2fbd5fb66be974debe032b89fe441db2eb1c9566fbc595babdf3cdc394e966a7","observation_id":"339c0bf8-6745-4aac-82de-512cda9eeb5b","resolution":{"observed_at":"2026-08-04T20:44:59.780638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.00410","last_updated":"2019-10-23T22:47:44Z","snapshot_observed_at":"2026-07-06T05:21:01.935928Z","submitted_at":"2016-12-01T20:12:40Z","title":"Deep Variational Information Bottleneck","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.00410","snapshot_observed_at":"2026-08-04T20:44:59.818605Z","title":"Deep variational information bottleneck.arXiv preprint arXiv:1612.00410, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T20:44:59.818605Z"},"links":{"cited_paper":"/paper/1612.00410","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:f8b658c531c6f9df86ecb33a4eab0a15eb27aad76ea43369b57e1127ae10fd59","observation_id":"176ac107-65d7-4870-88f5-82a4ff7d9531","resolution":{"observed_at":"2026-08-04T20:44:59.818605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.05453","last_updated":"2020-02-16T18:35:27Z","snapshot_observed_at":"2026-07-06T08:28:58.131087Z","submitted_at":"2019-10-12T00:55:06Z","title":"vq-wav2vec: Self-Supervised Learning of Discrete Speech Representations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.05453","snapshot_observed_at":"2026-08-04T20:44:59.929335Z","title":"vq- wav2vec: Self-supervised learning of discrete speech repre- sentations.arXiv preprint arXiv:1910.05453, 2019","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T20:44:59.929335Z"},"links":{"cited_paper":"/paper/1910.05453","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:f31f1b86e6f3e8a575619aaf0bfb5b3ac4fef34099355e987f3e422b22a2fd6f","observation_id":"f358c909-a9a3-40c7-8130-9d129917742f","resolution":{"observed_at":"2026-08-04T20:44:59.929335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.226728Z","title":"Con- trastively disentangled sequential variational autoencoder","venue":null,"work_id":"d2f6c8fb-233e-40c3-a944-0a29e15c09ad","year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.002204Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:900862de692ac9a59c041828c088ce3dfdfbe644ce23f1f2c0bd8bd72d62a625","observation_id":"c8754dd5-0333-4293-a1aa-ab733e27f642","resolution":{"observed_at":"2026-08-04T20:45:01.229668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17264","last_updated":"2023-03-30T10:01:58Z","snapshot_observed_at":"2026-08-06T04:48:48.866179Z","submitted_at":"2023-03-30T10:01:58Z","title":"Multifactor Sequential Disentanglement via Structured Koopman Autoencoders","version":1},"cited_work":{"arxiv_id":"2303.17264","doi":null,"metadata_source":"pith","pith_arxiv_id":"2303.17264","snapshot_observed_at":"2026-08-04T20:45:00.710486Z","title":"Multifactor Sequential Disentanglement via Structured Koopman Autoencoders","venue":"cs.LG","work_id":"f5d149df-97ee-44e9-903b-db7dbe3b7358","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.075667Z"},"links":{"cited_paper":"/paper/2303.17264","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:8c8588475256148dde81b913b86f08dcd7bb380bfa34515cfca3ce86ef495a3a","observation_id":"20e748f4-9b94-455f-bbad-d9bd9f825113","resolution":{"observed_at":"2026-08-04T20:45:00.713886Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2406.18131","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.697403Z","title":"Sequential disentanglement by extracting static information from a single sequence element.arXiv preprint arXiv:2406.18131, 2024","venue":null,"work_id":"5e1d2258-011d-4478-8440-0125c1c57d9a","year":2024},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.153908Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:9dac05f6e6b0d911a1d8823829f7c6e5a1aad075665a1f20fb1ce3e38ec05ca3","observation_id":"de86d914-4d0a-48dd-b670-10a2f1f27d10","resolution":{"observed_at":"2026-08-04T20:45:00.701740Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.217504Z","title":"Align your latents: High-resolution video synthesis with la- tent diffusion models","venue":null,"work_id":"e3c8cae5-ce83-4a84-afbf-15d65c045601","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.224278Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:41bc090f6967dbd63bc87f6446dae2f1692709c9f441b7ee6edd9a9fb762b3f1","observation_id":"73e9f7fd-319f-4328-bb41-a53466d4c0de","resolution":{"observed_at":"2026-08-04T20:45:01.220646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.208339Z","title":"Hyperreenact: One-shot reenactment via jointly learning to refine and re- target faces","venue":null,"work_id":"2f43142e-2d09-4e03-bdbc-3e2bd2a07368","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.298984Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:fe975cdb1499c2442d09f668db4c651bd87f8a5ed0456fa8db9cda9d8eeaac26","observation_id":"b3ce3522-5d8e-4090-837b-0088d8923cb3","resolution":{"observed_at":"2026-08-04T20:45:01.211345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.415445Z","title":"Video generation models as world simulators","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.415445Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:ebb4431c9199e7686ce15e75282564d98959c3085a8f39bed48597f61753e30c","observation_id":"70f5d5ea-f248-4cd3-b662-e7d52ff9eecf","resolution":{"observed_at":"2026-08-04T20:45:00.415445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.193633Z","title":"A simple framework for contrastive learning of visual representations","venue":null,"work_id":"266e0231-e867-4618-8e53-5995c49fe6e8","year":2020},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.418327Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:ab32a797bd766b81d891d626a0ff8f453ec8ccfba7c20a38b6ced9b47f3b4670","observation_id":"2bb72392-0bb5-4a9b-919c-2cef4aa34018","resolution":{"observed_at":"2026-08-04T20:45:01.196724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.421031Z","title":"John Wiley & Sons, 1999","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.421031Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:7ecef78ca0c7f52e8b0e9c3fee44f2c0f99b73e42c1d2da51eaf5618f888ad64","observation_id":"b87743df-5568-4186-bbb6-fb1839a28052","resolution":{"observed_at":"2026-08-04T20:45:00.421031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.179241Z","title":"Arcface: Additive angular margin loss for deep face recognition","venue":null,"work_id":"721980bd-9343-4800-ab4e-5a6f624b33a3","year":2019},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.423553Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:4b22d3b2301526b4c92ec8a3efe3d91159ce1eb513a05c465efc3a1cf3b65f2c","observation_id":"b715efbc-f76b-4b14-8f0c-1ac7e879d153","resolution":{"observed_at":"2026-08-04T20:45:01.182143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.141988Z","title":"Accurate 3d face reconstruction with weakly-supervised learning: From single image to image set","venue":null,"work_id":"179e8cd6-6aa5-471f-a892-6acde276f8e0","year":2019},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.425962Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:723ebde890b52dae079ab4d6640dba7207dd0abb43b10837dc4c99b847cdbcd2","observation_id":"4e9dfeb7-de44-489c-8ff6-a105ca0a81a9","resolution":{"observed_at":"2026-08-04T20:45:01.157518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-04T20:45:00.428335Z","title":"An image is worth 16x16 words: Trans- formers for image recognition at scale.arXiv preprint arXiv:2010.11929, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.428335Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:0a6f83efeea7deabd58cbfde420da149069a088e2bfbf8fed0d4fd1bdaa51028","observation_id":"fb3c9666-72f8-46a7-a219-1cc77241c2a4","resolution":{"observed_at":"2026-08-04T20:45:00.428335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.103230Z","title":"Headgan: One-shot neural head synthesis and editing","venue":null,"work_id":"a8dae226-554f-4ecf-a4ff-52eb81966a61","year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.431015Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:e0755edddb9035c8ad4ed73f6a40e1fdf6ee90bd2408d3cf98519f18b55ed87f","observation_id":"88c1bdb6-a236-43ee-abee-22ebefa460df","resolution":{"observed_at":"2026-08-04T20:45:01.115880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.433379Z","title":"Taming transformers for high-resolution image synthesis","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.433379Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:03e6d64dfa00730a72cae958462a5ee7848b9e4c29b12569775b55126c30b10a","observation_id":"a04ce27b-8f09-43d8-ab36-c0da17d58f8d","resolution":{"observed_at":"2026-08-04T20:45:00.433379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.056295Z","title":"High-fidelity and freely controllable talking head video generation","venue":null,"work_id":"21d5e6f7-0170-48d4-bad0-86155da7585b","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.436631Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:81f12d0c09e4d96f06433e499a89dfd8ea37f64d42b44e64bc82c92801eb4717","observation_id":"49aa3b0b-c009-4d71-ad6d-967eb51d2e41","resolution":{"observed_at":"2026-08-04T20:45:01.066815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.045270Z","title":"Implicit mo- tion function","venue":null,"work_id":"889c0a8c-1214-4593-ad9c-bd16ee9a6cb1","year":2024},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.439277Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:3e5c017a526207744e33fe2039762b054428aa72f8760bb9c02696a766849e0f","observation_id":"86275689-670c-4da0-a82c-7f96334106ed","resolution":{"observed_at":"2026-08-04T20:45:01.048407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.036188Z","title":"Generative adversarial nets.Advances in neural information processing systems, 27, 2014","venue":null,"work_id":"d2836cb5-9fd3-4daf-a8bd-debb19fe4e65","year":2014},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.441741Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:1ef48fb115b9dddf121b53bba2e1a638ec8354bff88cc92a70d12753ea3d29df","observation_id":"b0bc3d1d-9095-4073-99d7-3a1ab4fde155","resolution":{"observed_at":"2026-08-04T20:45:01.039310Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.027274Z","title":"Neu- ral head avatars from monocular rgb videos","venue":null,"work_id":"5a5626e0-3816-49bf-b736-ef8d543e8a23","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.444245Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:27e1436dd52ecd08709148b2b52fa11e0d013bdfb93a52954852c4ba34e8d9f7","observation_id":"6f7104ae-6eff-4679-b49b-90443825292c","resolution":{"observed_at":"2026-08-04T20:45:01.030206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.446697Z","title":"Bootstrap your own latent-a new approach to self-supervised learning.Advances in neural information processing systems, 33:21271–21284, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.446697Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:8597662c7df8166efd5f66084e4d86d08ff07d8f5ea9084f18ef28061f58c0cf","observation_id":"ff8efd54-27c6-452e-9ca7-8365de2220cd","resolution":{"observed_at":"2026-08-04T20:45:00.446697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:01.011391Z","title":"Vec- tor quantized diffusion model for text-to-image synthesis","venue":null,"work_id":"db1c9393-3bd5-44a8-8f96-7afb92c542ed","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.449095Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:f325ff12804a13e98497a660e6ac034c039722b5ad5c868045abd94a5b8bf908","observation_id":"44ade7db-6d7e-4c68-873a-71593b8151d2","resolution":{"observed_at":"2026-08-04T20:45:01.014558Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03168","last_updated":"2025-02-28T14:39:17Z","snapshot_observed_at":"2026-07-06T18:40:57.362549Z","submitted_at":"2024-07-03T14:41:39Z","title":"LivePortrait: Efficient Portrait Animation with Stitching and Retargeting Control","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03168","snapshot_observed_at":"2026-08-04T20:45:00.451507Z","title":"Livepor- trait: Efficient portrait animation with stitching and retarget- ing control.arXiv preprint arXiv:2407.03168, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.451507Z"},"links":{"cited_paper":"/paper/2407.03168","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:bfb7a4cb356c7f1b71153e9ca1970091a4c6ae05a3ef83fe7abc22e0d3f5f47a","observation_id":"bcf74821-70c3-4c65-b0f7-67f2497ff54f","resolution":{"observed_at":"2026-08-04T20:45:00.451507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.07496","last_updated":"2021-01-19T07:43:25Z","snapshot_observed_at":"2026-07-06T10:33:40.053591Z","submitted_at":"2021-01-19T07:43:25Z","title":"Disentangled Recurrent Wasserstein Autoencoder","version":1},"cited_work":{"arxiv_id":"2101.07496","doi":null,"metadata_source":"pith","pith_arxiv_id":"2101.07496","snapshot_observed_at":"2026-08-04T20:45:00.667918Z","title":"Disentangled Recurrent Wasserstein Autoencoder","venue":"cs.LG","work_id":"f7922602-a7e7-4f83-b9a7-03f5552174f7","year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.454436Z"},"links":{"cited_paper":"/paper/2101.07496","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:273bf309edff0c9b2c40ffeeb00792cf121c3b0802ff04799dd28afa7e032896","observation_id":"7ead3c4c-078f-4d97-a94e-7aa2969e32be","resolution":{"observed_at":"2026-08-04T20:45:00.671223Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.457205Z","title":"Momentum contrast for unsupervised visual rep- resentation learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.457205Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:a065a9968a0824feee5303e42955383864ed92da5e0a2b73b2c6bb47c0124ab1","observation_id":"b29780c0-0c41-4f32-a9f5-e455b60e919c","resolution":{"observed_at":"2026-08-04T20:45:00.457205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.459760Z","title":"Gans trained by a two time-scale update rule converge to a local nash equilib- rium.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.459760Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:e6e9729468ef37dbcb61beaf8758f55ea84f68dd11d15aedda1785092c027926","observation_id":"cc1637b3-7c5b-4428-85e8-7e30110622eb","resolution":{"observed_at":"2026-08-04T20:45:00.459760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.991685Z","title":"Look ma, no markers: holistic perfor- mance capture without the hassle.ACM Transactions on Graphics (TOG), 43(6), 2024","venue":null,"work_id":"8e0aa249-41d4-40aa-bf2c-23512e0105bc","year":2024},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.462180Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:d9e59ab85131177164b24eb02c99af9a3cd2396660270da604fbfd7666cbd9a9","observation_id":"b709d873-022d-4fcb-bab2-6a660130f0ab","resolution":{"observed_at":"2026-08-04T20:45:00.994554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.983233Z","title":"Denoising dif- fusion probabilistic models.Advances in neural information processing systems, 33:6840–6851, 2020","venue":null,"work_id":"201415a9-fd4e-4d86-b142-3b980083923d","year":2020},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.464875Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:f6534d3cde17412f3a74e142aba12dbb597ad71b71a090e0a7e426e29e546680","observation_id":"64d96769-e8ce-4192-8b4c-414f0c53c7fc","resolution":{"observed_at":"2026-08-04T20:45:00.986140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.974430Z","title":"Implicit identity representation conditioned memory compensation network for talking head video generation","venue":null,"work_id":"f756b9fe-3f41-4a40-bc1b-5a33f03fb184","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.467256Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:d30bb49a6d18b68c26c319b24e84a79c94a300d7622246593315fe5a725a7d49","observation_id":"94c082b6-9692-404e-8b0d-ff3f2410cad3","resolution":{"observed_at":"2026-08-04T20:45:00.977741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.965745Z","title":"Neu- ral compression-based feature learning for video restoration","venue":null,"work_id":"3ac3ae92-a9ed-43df-9a42-a82045119e10","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.469710Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:5f0ed78a88aa13126ff5e85ae70a03f375d32d8a285dbb300dd1cac7b8760148","observation_id":"9bd9ac8e-a9e5-4d05-9f0c-3d555b016df5","resolution":{"observed_at":"2026-08-04T20:45:00.968649Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17901","last_updated":"2023-11-29T18:53:34Z","snapshot_observed_at":"2026-07-06T16:54:33.882767Z","submitted_at":"2023-11-29T18:53:34Z","title":"SODA: Bottleneck Diffusion Models for Representation Learning","version":1},"cited_work":{"arxiv_id":"2311.17901","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.17901","snapshot_observed_at":"2026-08-04T20:45:00.653302Z","title":"SODA: Bottleneck Diffusion Models for Representation Learning","venue":"cs.CV","work_id":"c890812f-0272-456c-8b9d-0fad11d5af09","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.472120Z"},"links":{"cited_paper":"/paper/2311.17901","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:0cfab5b34a8649e98dfc3c2f03624042ccd62fb4afc8bb2335c0df830e73129c","observation_id":"3ed504cd-42f7-4e60-9833-97acbc02542e","resolution":{"observed_at":"2026-08-04T20:45:00.658366Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.956994Z","title":"Dis- entangled feature learning for real-time neural speech cod- ing","venue":null,"work_id":"23939d72-d39f-48d5-b219-6e1d2708f2a3","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.474731Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:32f9c382274565ed3222db36b7f1648f5f8033a3d4d6a768a3e59a5db22c5bc8","observation_id":"48b15062-8264-467c-821b-010c7e25f131","resolution":{"observed_at":"2026-08-04T20:45:00.959832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.947532Z","title":"Elucidating the design space of diffusion-based generative models.Advances in Neural Information Processing Sys- tems, 35:26565–26577, 2022","venue":null,"work_id":"685e7a47-dbd3-4bcd-8efb-0d437f831d06","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.477767Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:8de242a98149ad19122d657da20518d9e0ab1912b4453e57d7a07067df0a0134","observation_id":"bbbdfc4e-e55d-40d0-a69b-fa9f73034fcc","resolution":{"observed_at":"2026-08-04T20:45:00.950778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.938516Z","title":"Deep contextual video com- pression.Advances in Neural Information Processing Sys- tems, 34:18114–18125, 2021","venue":null,"work_id":"e87e646f-0970-4a39-9125-36e294e256d5","year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.480229Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:4ccc90af43ca18a4d1145f03317319804a0c13e815b2879161fe62d4aa494bb1","observation_id":"1beb5af6-5393-4386-ab3d-b7f890b1aa37","resolution":{"observed_at":"2026-08-04T20:45:00.941726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.929762Z","title":"Hybrid spatial-temporal en- tropy modelling for neural video compression","venue":null,"work_id":"4e69ee99-dcb0-4ee7-bc45-68ba06bb2518","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.482629Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:ddb4f1371fff52169db61fd2493d9fab4b5bdb7473fb1066ce6fff1b6b7e3706","observation_id":"0647afbd-ee36-4294-a2e9-db02df80d2a2","resolution":{"observed_at":"2026-08-04T20:45:00.932611Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.920398Z","title":"Neural video compression with diverse contexts","venue":null,"work_id":"30bac177-ca93-4060-a8ea-65d1c5f31d55","year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.485089Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:e6397d684ab49d87d35fb9637c8287e50fec0f6dfaeaea82e6a21c679b82c470","observation_id":"77b4db0a-4aca-41d5-a7ad-1549fc582590","resolution":{"observed_at":"2026-08-04T20:45:00.923240Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-04T20:45:00.487487Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models.arXiv preprint arXiv:2301.12597, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.487487Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:e8fdbfe62c85b0cb1b3aa939b311724f36cd3e24baf98c5e66447715f4525eba","observation_id":"8db2abca-719e-4ffa-a2f8-3a41cd3ed4df","resolution":{"observed_at":"2026-08-04T20:45:00.487487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.911669Z","title":"Motion-focused contrastive learning of video representations","venue":null,"work_id":"6362ddd3-412f-429c-b184-56614ca3b606","year":null},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.490123Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:ddd9f0c7e1055dcd2169c750f902681dc1a5130ddb5bd6722662194b099126dd","observation_id":"daff29dc-a889-453b-8aff-de3e3b308a8d","resolution":{"observed_at":"2026-08-04T20:45:00.914474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03701","last_updated":"2024-11-01T14:48:57Z","snapshot_observed_at":"2026-07-06T16:57:51.263433Z","submitted_at":"2023-12-06T18:59:31Z","title":"Return of Unconditional Generation: A Self-supervised Representation Generation Method","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03701","snapshot_observed_at":"2026-08-04T20:45:00.492768Z","title":"Self- conditioned image generation via generating representations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.492768Z"},"links":{"cited_paper":"/paper/2312.03701","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:bf575e37f8f6f24e8e98e7bdc14858a44f74bf252abdbc6e7a2822cec1335249","observation_id":"209ba1d1-a78d-4e50-a545-22f8de9a3053","resolution":{"observed_at":"2026-08-04T20:45:00.492768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.02991","last_updated":"2018-06-12T17:18:55Z","snapshot_observed_at":"2026-07-06T06:27:12.758357Z","submitted_at":"2018-03-08T07:42:39Z","title":"Disentangled Sequential Autoencoder","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.02991","snapshot_observed_at":"2026-08-04T20:45:00.495406Z","title":"Disentangled sequential autoencoder.arXiv preprint arXiv:1803.02991, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.495406Z"},"links":{"cited_paper":"/paper/1803.02991","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:a7c36e1452ba80bffbb73612560065f31f0092345cc87760764eb0111868ae91","observation_id":"6983b08e-cd67-4bab-b698-8fee569afcba","resolution":{"observed_at":"2026-08-04T20:45:00.495406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.902718Z","title":"Anitalker: animate vivid and di- verse talking faces through identity-decoupled facial motion encoding","venue":null,"work_id":"ddf93bd6-7097-4512-9486-3374224c76a6","year":2024},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.498188Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:76d4785a07972eb4b448e5342c4c4705df4ce7d81ce164770d61d836feabe7de","observation_id":"c087c444-9eb8-41a3-b8a4-f48e1bf50842","resolution":{"observed_at":"2026-08-04T20:45:00.905789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-04T20:45:00.500592Z","title":"Decoupled weight decay regularization.arXiv preprint arXiv:1711.05101, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.500592Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:618c8e48f16496a33c1c40542d57ed8a608aec913a852e6aa909679f9b5f89a2","observation_id":"46a7caf5-e4e9-4048-b0b2-df716ddefb6a","resolution":{"observed_at":"2026-08-04T20:45:00.500592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.893827Z","title":"Implicit warping for animation with image sets.Advances in Neural Information Processing Systems, 35:22438–22450, 2022","venue":null,"work_id":"926ec0aa-71a5-408f-bd93-0c59d6dc32f0","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.502984Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:354bee81503f0e6f9316065e201c2183f46e614ee0c22372ea093852fef77c3f","observation_id":"2961290c-86bc-4dfb-bb57-e2d2afee2176","resolution":{"observed_at":"2026-08-04T20:45:00.896743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.885102Z","title":"Sample and predict your latent: modality-free sequential disentan- glement via contrastive estimation","venue":null,"work_id":"381b2b2e-c016-486c-87d3-03687a053051","year":null},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.505418Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:ea0052f2805e66095c7ab894b2d023a8fd17ffa36561f3c97fa159a006d8c57d","observation_id":"1a5cc352-2b5e-4a1b-9eb5-47416e825300","resolution":{"observed_at":"2026-08-04T20:45:00.887939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.507990Z","title":"Scalable diffusion models with transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.507990Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:d130004152e0fd60e045a76c5a50de35e80758cfea120f4bafd6c9cc91514785","observation_id":"4cf92cb0-e99f-4eea-90d8-d530ade0e742","resolution":{"observed_at":"2026-08-04T20:45:00.507990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2208.06366","last_updated":"2022-10-03T11:47:10Z","snapshot_observed_at":"2026-08-04T18:14:09.239336Z","submitted_at":"2022-08-12T16:48:10Z","title":"BEiT v2: Masked Image Modeling with Vector-Quantized Visual Tokenizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.06366","snapshot_observed_at":"2026-08-04T20:45:00.510559Z","title":"Beit v2: Masked image modeling with vector-quantized visual tokenizers.arXiv preprint arXiv:2208.06366, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.510559Z"},"links":{"cited_paper":"/paper/2208.06366","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:9bc3a093bf982d63fd4c1834148df506a32d22c10697d251bd9a177a9194369d","observation_id":"bccd00af-3f1f-4671-abb2-1cbe3714043b","resolution":{"observed_at":"2026-08-04T20:45:00.510559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.870212Z","title":"Language models are unsu- pervised multitask learners.OpenAI blog, 1(8):9, 2019","venue":null,"work_id":"36c086d9-f326-4480-9750-c690af4f21b8","year":2019},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.513304Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:eca8153fd8fd714ca72401c65f518dc0f2dc4fcd35be544ec756656393f3f565","observation_id":"b3b02db2-c1a0-43d9-b520-0c1210ba4ca7","resolution":{"observed_at":"2026-08-04T20:45:00.873180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.515665Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.515665Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:41cc34f5a5bd1f3c034e59dfbbdd15f91514de46647ebfb4c0f460a661bc004f","observation_id":"5bc423be-4c77-4d2f-a436-9e3dc23ac74f","resolution":{"observed_at":"2026-08-04T20:45:00.515665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.854072Z","title":"Deep visual analogy-making.Advances in neural informa- tion processing systems, 28, 2015","venue":null,"work_id":"9e4d6b4e-3767-4760-a3c8-eb5b460b1a8d","year":2015},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.518294Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:22913f54b000fdcaba38fb98f6d5ae336521e8408e57e89077f0780463a936e8","observation_id":"674dd3b9-d95f-4808-b7c0-ed6286fe2915","resolution":{"observed_at":"2026-08-04T20:45:00.856960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.520776Z","title":"High-resolution image synthesis with latent diffusion models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.520776Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:297e2758f1ab58ac01599dd2717c94933f3ddd7c14208c216ea68617c4fca954","observation_id":"1e3365b7-6ffa-4f63-ba65-c965aef37c9d","resolution":{"observed_at":"2026-08-04T20:45:00.520776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.838818Z","title":"Temporal context mining for learned video compression","venue":null,"work_id":"b42fa850-a26e-44d3-8383-9574039cbec3","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.523096Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:55ea359bdeb550b5ccd0d217fd36211ac34f544a3cfbad6992cc8f7df47fb825","observation_id":"5860d4c5-2bfb-4557-ab95-6b51ee07dc49","resolution":{"observed_at":"2026-08-04T20:45:00.841937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.830016Z","title":"First order motion model for image animation.Advances in neural information processing systems, 32, 2019","venue":null,"work_id":"fe2c210f-37af-434a-9c1b-ade027097d03","year":2019},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.525513Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:218593e5188ceb31fcb1f689c53a1318a09966f7a87d41ac7b02d758dd025230","observation_id":"5f425ff5-c232-4608-96bb-c7eea3a9a316","resolution":{"observed_at":"2026-08-04T20:45:00.832906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.821204Z","title":"Sequential representation learning via static- dynamic conditional disentanglement","venue":null,"work_id":"b3f291ab-d972-48d0-bb89-de437d91f937","year":2024},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.527999Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:bdf6d9e2c4ca9b601eb657ba15a53a88b0d2b69b37e17eb79cd41d87525517b7","observation_id":"4696cc0b-b4d9-4315-b298-6c1caf427c42","resolution":{"observed_at":"2026-08-04T20:45:00.824126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"physics/0004057","last_updated":"2000-04-24T15:22:30Z","snapshot_observed_at":"2026-07-07T06:44:19.645435Z","submitted_at":"2000-04-24T15:22:30Z","title":"The information bottleneck method","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"physics/0004057","snapshot_observed_at":"2026-08-04T20:45:00.530551Z","title":"The information bottleneck method.arXiv preprint physics/0004057, 2000","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.530551Z"},"links":{"cited_paper":"/paper/physics/0004057","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:54cdb9d275b47b0ab68afbeda1c42e03d6e0c2d04c646335c065ec55890f8116","observation_id":"fb1e53ab-5faf-40eb-a054-d59259d2bb16","resolution":{"observed_at":"2026-08-04T20:45:00.530551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.533176Z","title":"Neural discrete representation learning.Advances in neural information pro- cessing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.533176Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:54d021df1cda1f6c7b6d2b5b1860fe6157ccb716be3b8d679dd836fa2412ad25","observation_id":"f0233383-231b-422f-b2c6-26a17a180104","resolution":{"observed_at":"2026-08-04T20:45:00.533176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.535702Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.535702Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:8380b36533324071a1dc0f257e9b841a5e500eea63189a4aed6fac9dd4dfa774","observation_id":"1fcc528c-460b-4999-bc8e-573478ae28a0","resolution":{"observed_at":"2026-08-04T20:45:00.535702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.800515Z","title":"One-shot free-view neural talking-head synthesis for video conferenc- ing","venue":null,"work_id":"5becf19f-75ea-44da-b578-f9201498ee03","year":null},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.538306Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:113b6eefe3d6ccb2e59db1f770bbf009c701acd078005d8c7fd651a10539c754","observation_id":"aa47434a-3805-47b8-a268-411b6ddb71bc","resolution":{"observed_at":"2026-08-04T20:45:00.803471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.791069Z","title":"Latent image animator: Learning to animate im- ages via latent space navigation","venue":null,"work_id":"a96821f8-868e-4dd5-aebc-6eccb6af07af","year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.540946Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:ee9500324e9480d63cae7b3d3f90237f373d096824ccbac035aad6de9648a4ce","observation_id":"a5a988b9-9fd0-4478-b2c3-d71c924ba1c3","resolution":{"observed_at":"2026-08-04T20:45:00.794245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.03771","last_updated":"2020-07-14T03:42:34Z","snapshot_observed_at":"2026-07-06T08:27:58.343233Z","submitted_at":"2019-10-09T03:23:22Z","title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.03771","snapshot_observed_at":"2026-08-04T20:45:00.543485Z","title":"Huggingface’s transformers: State-of-the-art natural language processing","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.543485Z"},"links":{"cited_paper":"/paper/1910.03771","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:2dac49681dbdaff15fe25a788777f37b8a8bd69796f2c509f25d631ea36f08c1","observation_id":"5124988f-ff46-4de7-91b5-3b70bb00a2d0","resolution":{"observed_at":"2026-08-04T20:45:00.543485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.781743Z","title":"Fake it till you make it: face analysis in the wild using synthetic data alone","venue":null,"work_id":"a3adc8de-22a8-408f-be16-65ef580145a9","year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.546104Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:84de87aa1fefc2bdf665ba8cbf582cecded83a3f4d779dca05002a5271df53b4","observation_id":"54cf12a0-594e-42a3-91f4-5092bc7bf123","resolution":{"observed_at":"2026-08-04T20:45:00.784700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.772327Z","title":"3d face reconstruction with dense landmarks","venue":null,"work_id":"e8507e98-b71f-4945-bf1e-e9b31cd71715","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.548524Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:6099be69a93fa86ebb9caefda6af0f4c6bdf41291204c5f1e267e2f9cc6d185b","observation_id":"824f66db-771b-4ec6-8bf0-586e645ffb66","resolution":{"observed_at":"2026-08-04T20:45:00.775361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.763286Z","title":"N ¨uwa: Visual synthesis pre- training for neural visual world creation","venue":null,"work_id":"bf1e1c76-cfbb-4d18-99bf-1798ffe19617","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.551119Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:8de6a5ffaae0e0cd34bc0e6f1706626dd5837a387e37a0db36e4be43f4aeb918","observation_id":"d1e3b02f-0a02-47b6-9958-b33230548775","resolution":{"observed_at":"2026-08-04T20:45:00.766313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.754050Z","title":"Simmim: A simple framework for masked image modeling","venue":null,"work_id":"a1fcde86-975d-4262-a7d1-39a6b172c070","year":2022},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.553618Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:05c8a6e5cba1b772a3a1ea22d8c6a8652fbd68bbfab9eea02b2310a6d1089024","observation_id":"d1743a51-6937-4a2c-9b35-75071861155c","resolution":{"observed_at":"2026-08-04T20:45:00.756997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.10157","last_updated":"2021-09-14T21:20:06Z","snapshot_observed_at":"2026-07-06T11:02:11.424356Z","submitted_at":"2021-04-20T17:58:03Z","title":"VideoGPT: Video Generation using VQ-VAE and Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.10157","snapshot_observed_at":"2026-08-04T20:45:00.556081Z","title":"Videogpt: Video generation using vq-vae and trans- formers.arXiv preprint arXiv:2104.10157, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.556081Z"},"links":{"cited_paper":"/paper/2104.10157","citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:7c66ffcf9ab031e3bdb42855abf1cc14da2997895b34e7b5c3151ba5af4374f5","observation_id":"566023bf-bb8c-44a6-95c0-bcc90c273c38","resolution":{"observed_at":"2026-08-04T20:45:00.556081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T20:45:00.744652Z","title":"S3vae: Self-supervised sequential vae for representation disentanglement and data generation","venue":null,"work_id":"d4096860-c0d8-4327-ab85-c6ec4626e528","year":2020},"citing_paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-04T20:45:00.558941Z"},"links":{"citing_paper":"/paper/2509.08376"},"observation_digest":"sha256:6332c1a7cce1e8a14b2450617e1922584cc08cc93af49569c61f87c110d1b867","observation_id":"adf97a62-aeaa-42a2-86dd-17dcf09a051d","resolution":{"observed_at":"2026-08-04T20:45:00.747761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.08376","last_updated":"2025-09-10T08:14:45Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-05T17:27:40.320405Z","submitted_at":"2025-09-10T08:14:45Z","title":"Bitrate-Controlled Diffusion for Disentangling Motion and Content in Video"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":4,"verified_fuzzy":38},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 0 inbound Pith citation observations for arXiv:2509.08376."}