{"as_of":"2026-08-22T09:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:95a2ffa325433fcaf84f813a29bc27d1ac3ba3d4ef30fc61f7394ad49446e9da","coverage":[{"denominator":61,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":61,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T15:42:33.558286Z","state":"measured"},{"denominator":61,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":61,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.10778/citation-record","integrity":"/paper/2412.10778/integrity","json":"/paper/2412.10778/citation-record.json","paper":"/paper/2412.10778"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.224078Z","title":"Human-level control through deep reinforcement learning,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.224078Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:b3622d0830246dff7294f3c253de1284c93a1758b1baba6d3c22152d248343c1","observation_id":"5ed2e75b-077b-4876-949a-a62e9bc3093b","resolution":{"observed_at":"2026-08-11T15:42:33.224078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.790397Z","title":"Continuous control with deep reinforcement learning,","venue":null,"work_id":"4849d883-102d-4885-bf41-ed023acf2d4c","year":2016},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.230030Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:f0eec9d6cd804146d75c6fbe43262ab60f1c2ebc4c6518734605b87c86ab9dfc","observation_id":"75f78713-3e70-4d4d-ad11-82bdfb58d759","resolution":{"observed_at":"2026-08-11T15:42:34.796402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.772241Z","title":"Balancing state exploration and skill diversity in unsupervised skill discovery,","venue":null,"work_id":"7ec20bb1-bdbd-4983-a549-3db2b00274f5","year":2025},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.235252Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:2d8ae98bc810da0a4ae12cc6224f75178235642a5cb610a51c1626794837529d","observation_id":"5de010b6-9673-49dd-9151-831ae5e3ea94","resolution":{"observed_at":"2026-08-11T15:42:34.778580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.752678Z","title":"URLB: Unsupervised reinforcement learning benchmark,","venue":null,"work_id":"d192550f-283d-4162-9c5f-cec70643c26b","year":2021},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.240468Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:b3417139bb8b735c0fa7606382a67d236bbeb59aac1bb3325a8802de9a379031","observation_id":"ac017a37-d82b-47ef-92a8-4c2010a2d2ff","resolution":{"observed_at":"2026-08-11T15:42:34.758897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.733178Z","title":"Effective representation learning is more effective in reinforcement learning than you think,","venue":null,"work_id":"be1e1e4b-7a7a-47e8-b5aa-4fc1efa5be5b","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.245565Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:619e3e7871001ca76cf06b6ecaf858e30637758ff702e4b71c1dccc7f60bd8c0","observation_id":"c61a4e09-b102-4216-95f2-862dd6175a56","resolution":{"observed_at":"2026-08-11T15:42:34.738619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.715740Z","title":"Taco: Temporal latent action-driven contrastive loss for visual reinforcement learning,","venue":null,"work_id":"3d7eac9b-c1fa-4007-8c6d-2b713bd35bc5","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.250551Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:d5b1cf1fd868e7029152951a8d41788d9ab5a2e1e628080d693f0c5b44148adc","observation_id":"99a9fef7-e0e0-4b70-aa28-ed4e90520bc3","resolution":{"observed_at":"2026-08-11T15:42:34.721312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.697629Z","title":"Human-level control through directly trained deep spiking q-networks,","venue":null,"work_id":"c09fad04-9b31-4ce3-988c-1c385244c25c","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.256568Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:f50474104987fa209573316832e0f66727f4f29f3b4aa00fbad5d0ca0f0e5673","observation_id":"2b3bdfb8-989a-4609-ba87-799b0c30edd5","resolution":{"observed_at":"2026-08-11T15:42:34.703923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.677227Z","title":"Deep reinforcement learning-based automatic exploration for navigation in unknown environment,","venue":null,"work_id":"418fe720-0a8e-4e54-9ae0-79bdf944a7f6","year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.261642Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:368c4fedd9773b4f52bec0192f5fcc458206b1664a2734d0630f27f51dfb5f18","observation_id":"cb5f13de-c83b-44ee-99f3-f9cdedcfb322","resolution":{"observed_at":"2026-08-11T15:42:34.684276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.657000Z","title":"Model based reinforcement learning for atari,","venue":null,"work_id":"0df46334-d239-432b-b1c3-2a120fe8a8f9","year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.266758Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:460c55b08c5f2947b49e7f5efbe75880ae72f17612cad0ff53ca662e27a3d27d","observation_id":"0f39e2bf-452a-46fe-bc25-7d06a25b6d90","resolution":{"observed_at":"2026-08-11T15:42:34.663078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.271917Z","title":"Dream to control: Learning behaviors by latent imagination,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.271917Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:e05b842da0ce67b75e3ebd6c7a0ff065cde7d7dcc22c581fb6554b14a7823a14","observation_id":"b51610b5-ad8c-45ed-bbbb-7193d00240ae","resolution":{"observed_at":"2026-08-11T15:42:33.271917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.624387Z","title":"Prototypical context-aware dynamics for gener- alization in visual control with model-based reinforcement learning,","venue":null,"work_id":"92ceebf1-94b3-4e47-853d-54857401251e","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.277054Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:2aa8020285b041eaf24af34ee3f23efeebb22e53e00ee9b1cb5957c6e6d32ac4","observation_id":"7150e8d8-ec79-44fc-9112-9d6d96fe8901","resolution":{"observed_at":"2026-08-11T15:42:34.631752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.283182Z","title":"Mastering atari with discrete world models,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.283182Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:123fd6d2f64169c69778e31cfa64a64e9ffc5da54e24d8ec01283af2c753ad7b","observation_id":"b45358da-3250-4f90-a234-355855e0d2dd","resolution":{"observed_at":"2026-08-11T15:42:33.283182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.594106Z","title":"Data-efficient reinforcement learning with self- predictive representations,","venue":null,"work_id":"2593de1b-4ee6-40ba-9956-2a970e4a9231","year":2021},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.289176Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:c8189f6b5f143ebbc51b0048f62c9942114b47615e62271766123b0ab4db4b31","observation_id":"72b4e295-8987-4040-a883-3dccd86a2545","resolution":{"observed_at":"2026-08-11T15:42:34.600499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.574949Z","title":"Reinforcement learning with unsupervised auxiliary tasks,","venue":null,"work_id":"ec2a98cf-b701-4c70-965d-31ae13b86f3a","year":2017},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.294585Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:845fb9ce222a537ff982ea95402156b81115f3948764fb5d2e89cdbfdcc915c7","observation_id":"75c57e4c-0eab-42c1-9946-51aad1536342","resolution":{"observed_at":"2026-08-11T15:42:34.580243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.557748Z","title":"Masked and inverse dynamics modeling for data-efficient reinforcement learning,","venue":null,"work_id":"02450e4f-5ff0-47d1-9fdd-f26b64e41de5","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.300543Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:c887eb3b26aeff90938b45b3bc392ceb5a4a4956ba842cd87ad8b57110f5a93f","observation_id":"945fdef8-df89-4e98-b6ef-20ae0bf5b9ee","resolution":{"observed_at":"2026-08-11T15:42:34.563656Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.540343Z","title":"Learning future representation with synthetic observations for sample-efficient reinforcement learning,","venue":null,"work_id":"81d34a54-287a-482c-89f4-5259693c899d","year":2025},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.306734Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:b6ee560aaf896d031ebfda0b0355bbb292eac7571a68c31a5b6f58cb1ced16fe","observation_id":"2e7ae6f5-b4de-4792-aa5d-2619cbcb54d2","resolution":{"observed_at":"2026-08-11T15:42:34.545905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.524122Z","title":"Design from policies: Conservative test-time adaptation for offline policy optimization,","venue":null,"work_id":"987feec8-d800-4639-af63-1494915033a5","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.312306Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:e3de281fc39bcad5923107b2aedc1a5a2175f51779640831b11f49516149641b","observation_id":"c1c80603-0cf8-4ecf-afc9-1a341a610239","resolution":{"observed_at":"2026-08-11T15:42:34.529294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.507365Z","title":"Hiql: Offline goal-conditioned rl with latent states as actions,","venue":null,"work_id":"7f4c4a88-3d39-4407-81b1-eb97d0317d1c","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.318820Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:6e6fd9e7c6134e0bf28cc63babbe9b31cf166ca260aa105a6f970d4a4f15075c","observation_id":"2ebd97ab-c429-4b4a-8061-1996a4e9d893","resolution":{"observed_at":"2026-08-11T15:42:34.513012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.324977Z","title":"A survey of imitation learning: Algorithms, recent developments, and challenges,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.324977Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:646975abf927743fc4d47e03816c4e97f113aca26533d65c4f8aae7af9b78926","observation_id":"dcd66665-e15e-4580-8347-2062c76fe5f2","resolution":{"observed_at":"2026-08-11T15:42:33.324977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.477711Z","title":"Generative adversarial imitation learning,","venue":null,"work_id":"6b38d68c-2609-4250-ae32-a8dc41aa94ab","year":2016},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.330351Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:185a28e551821d728fc6cb48459f29f483e7da5312b82097d929c076610b22cd","observation_id":"7a86abef-1e76-47f0-9d19-173f8d81fdcd","resolution":{"observed_at":"2026-08-11T15:42:34.483974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.457961Z","title":"Robotic offline rl from inter- net videos via value-function learning,","venue":null,"work_id":"ee519e8d-7ccc-4b80-a32b-6317b3e887b4","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.335615Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:65a76cb2c74cca2fe652d4c4d2d8dbe5bf3d731df4ec1d954f19e6a75b6ca7f8","observation_id":"50fd8ef2-f333-43b3-8f10-d4f093d43eaf","resolution":{"observed_at":"2026-08-11T15:42:34.464282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.434394Z","title":"Reinforcement learning from passive data via latent intentions,","venue":null,"work_id":"1f021e51-7e37-4227-be92-58c43ece20b4","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.341267Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:6e053324478db8093bc1f0c484871305f9b76ba002f8a944c39190cb4142caa3","observation_id":"2b8cb51b-625e-4a30-9037-8501e69ed6a6","resolution":{"observed_at":"2026-08-11T15:42:34.440902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.414739Z","title":"Diffusion reward: Learning rewards via conditional video diffusion,","venue":null,"work_id":"31724cd3-2175-4c78-bd8a-cdf52636f7b5","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.347723Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:77ac21129eda58ea8da7093eaeb43bbdce4ba7ccf73cb877c324a1623db24ba1","observation_id":"0bfce7e9-df19-42b7-9c4c-25b243d89072","resolution":{"observed_at":"2026-08-11T15:42:34.421039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.06158","last_updated":"2019-06-18T04:56:56Z","snapshot_observed_at":"2026-08-18T08:58:09.128606Z","submitted_at":"2018-07-17T00:25:15Z","title":"Generative Adversarial Imitation from Observation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.06158","snapshot_observed_at":"2026-08-11T15:42:33.353982Z","title":"Generative adversarial imitation from observation,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.353982Z"},"links":{"cited_paper":"/paper/1807.06158","citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:b08f8f6cd3851faf710f62ec90d71bce0a836b4e5a925de1eb50924a841bdc90","observation_id":"a839e7bd-7fc0-4c02-aa5f-dad5ca4ca3a5","resolution":{"observed_at":"2026-08-11T15:42:33.353982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.389610Z","title":"Learning from visual observation via offline pretrained state-to-go transformer,","venue":null,"work_id":"bec8335b-23f8-46c1-ad2d-0555644496a3","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.359366Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:4f4676b84b4d671b24621d172fbd9142702bed56ada29c97d970e692a3edc271","observation_id":"d87404cf-9816-436c-83be-3c39d3fd6826","resolution":{"observed_at":"2026-08-11T15:42:34.395230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.369042Z","title":"Video prediction models as rewards for reinforcement learning,","venue":null,"work_id":"414a8d50-2ffe-447c-aa64-82b577a4ad39","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.364350Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:12ed6ccfd25cc0089224b4f1d405cd37a2aec1cc9e2858033f4c4f83efa5556c","observation_id":"0d4c5766-18cf-4014-8f9c-8721bb100b63","resolution":{"observed_at":"2026-08-11T15:42:34.375820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.344292Z","title":"Ilpo-mp: Mode priors prevent mode collapse when imitating latent policies from observations,","venue":null,"work_id":"5862221d-a569-4c26-9e44-4cd04d4b4707","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.369425Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:31a4decf63e31777324a9f610e95f8540b4c160b55a83a183a17231f8bb362e9","observation_id":"1e5283bb-5ed2-4842-86c4-729eae8464fc","resolution":{"observed_at":"2026-08-11T15:42:34.351401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.374553Z","title":"Behavioral cloning from obser- vation,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.374553Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:26ae6acc94e5a625ed39cebeef07699c04b2e41d6a90cd45ccc9b3925570780e","observation_id":"017d5ac7-d16f-40da-aaa4-5a5907a83b75","resolution":{"observed_at":"2026-08-11T15:42:33.374553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.306646Z","title":"Imitating latent policies from observation,","venue":null,"work_id":"1b8e1f56-1cb8-4712-97d4-7541aa1c75f6","year":2019},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.379850Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:bb7fdf09e9d99da2b5851c4b4567a764057f63729cd0c2618fec0d7cfff55494","observation_id":"d97d5b92-0e4b-4693-bdee-484552afcb7b","resolution":{"observed_at":"2026-08-11T15:42:34.313740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.281674Z","title":"Steps: Joint self-supervised nighttime image enhancement and depth estimation,","venue":null,"work_id":"db40b667-9438-4503-8a78-c28cc98fb8b0","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.386170Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:d0071460f88b7ce8c1dc4a10ca7b8b4b571f7fc75fd0b0aad3511c783cffd63f","observation_id":"2a777e21-585c-44ee-a9de-6df24ef2f5d4","resolution":{"observed_at":"2026-08-11T15:42:34.288541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.392657Z","title":"Reinforcement learn- ing with prototypical representations,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.392657Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:1419ffe35907369c3017db76dba4de14506d80762abbe8f98b5970ae310ad8a4","observation_id":"eb839d9c-41d2-4a90-aec5-54b49744cb68","resolution":{"observed_at":"2026-08-11T15:42:33.392657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.245204Z","title":"Intrinsically motivated self- supervised learning in reinforcement learning,","venue":null,"work_id":"aeed98eb-7110-4886-a2f5-826899efec0e","year":2022},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.398455Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:4887acf7a234bd4ddb9e8f033d8aabac107ead18856d2f4c5aa18f7b466b0917","observation_id":"ea9ce33c-feff-4c41-a8f6-5b3773b56d2a","resolution":{"observed_at":"2026-08-11T15:42:34.251760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.403630Z","title":"Deep reinforcement learning for autonomous driving: A survey,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.403630Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:9d860c4edef74908a8b2445546844c2130b43561ffd7e6ba77b06e6cad74e224","observation_id":"23458c74-1c8a-4f5c-9e81-c362bf614def","resolution":{"observed_at":"2026-08-11T15:42:33.403630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.206902Z","title":"Scalable deep reinforcement learning for vision-based robotic manipulation,","venue":null,"work_id":"9618e553-f094-4951-84c8-ef3da17317a2","year":2018},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.409028Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:02b56180b461fc4025cfe2150cb5a3215ecb25e67ac17cb840ecb02df10edb33","observation_id":"5ab1d93a-620a-4ff1-bfd7-04c8e921c0a6","resolution":{"observed_at":"2026-08-11T15:42:34.216852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04104","last_updated":"2024-04-17T17:41:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-01-10T18:12:16Z","title":"Mastering Diverse Domains through World Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04104","snapshot_observed_at":"2026-08-11T15:42:33.413904Z","title":"Mastering diverse domains through world models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.413904Z"},"links":{"cited_paper":"/paper/2301.04104","citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:d6944d2bc731eed6c503bbd20e0782f68bec96bd6cdda74beae7ec321f5cf19f","observation_id":"aef02e9f-444a-4137-b4bb-4f2cc325b220","resolution":{"observed_at":"2026-08-11T15:42:33.413904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.172746Z","title":"Dreamerpro: Reconstruction-free model-based reinforcement learning with prototypical representa- tions,","venue":null,"work_id":"b213cb23-49d4-40bc-b617-d8f74634feeb","year":2022},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.419892Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:c5556514f4439a5ff4b10942b601a408260514bd616249b0c6e7c16b67a637c9","observation_id":"4436880e-2654-4eb8-b6ed-3ad56388f8e0","resolution":{"observed_at":"2026-08-11T15:42:34.180883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.153183Z","title":"A survey on model-based reinforcement learning,","venue":null,"work_id":"931a1d80-d1f6-480c-94ff-318386c616b1","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.425363Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:e3be25a74587741f3f5d2ce6541a4f46f44c86e0f6a4b870c73372df51dfb26c","observation_id":"3356251f-21b9-46b5-82a0-1edae7ee10ea","resolution":{"observed_at":"2026-08-11T15:42:34.158970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.135361Z","title":"Curl: Contrastive unsupervised representations for reinforcement learning,","venue":null,"work_id":"d97eae7a-ce8c-4050-9af9-6ad40e9a167e","year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.430342Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:5a461e2bd8cc75616662d587fc809ede9bb64e4693c9b20fda8a14edfc45e72f","observation_id":"78ea1545-2bd2-4a86-afaa-0b9730b4e47a","resolution":{"observed_at":"2026-08-11T15:42:34.140516Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.06173","last_updated":"2022-03-11T18:58:10Z","snapshot_observed_at":"2026-08-16T17:15:14.047305Z","submitted_at":"2022-03-11T18:58:10Z","title":"Masked Visual Pre-training for Motor Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.06173","snapshot_observed_at":"2026-08-11T15:42:33.435199Z","title":"Masked visual pre- training for motor control,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.435199Z"},"links":{"cited_paper":"/paper/2203.06173","citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:3299b34b03aa6a145639a2c26963c5c0df28ce150e7b3761d12a0a1a16511260","observation_id":"83f10578-17bc-4f3f-af4b-2dad5d55dcf2","resolution":{"observed_at":"2026-08-11T15:42:33.435199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.117589Z","title":"Value-consistent representation learning for data-efficient reinforcement learning,","venue":null,"work_id":"b7cd8536-d080-4c16-8ec2-d5118d6fcf99","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.440609Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:80d2e5fabc2b9d9635ccbc2124ddb1ac6651180fd77fcac620fada82b1d940af","observation_id":"5242b6dd-189d-42cb-89f9-d7d51ba8f1e2","resolution":{"observed_at":"2026-08-11T15:42:34.123274Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.098117Z","title":"Cross-domain random pretraining with prototypes for reinforcement learning,","venue":null,"work_id":"075d603e-2dd5-4fc2-a873-73bee6b486f4","year":2025},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.445845Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:4208b3b9cfa91dfc0d6704f7abc615d1dace8e98340417bd082eee38a9382cbf","observation_id":"20d17f4e-42d7-496a-b606-8ba878d5f664","resolution":{"observed_at":"2026-08-11T15:42:34.104164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.451288Z","title":"Imitation learning: A survey of learning methods,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.451288Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:dd6c75a89e195cfd1c96b302978345e148b065bb8940e205921ea31c9e5f2dba","observation_id":"2a8ee257-08a9-41b3-9b97-6738bc967c98","resolution":{"observed_at":"2026-08-11T15:42:33.451288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.066235Z","title":"Reinforcement learning with action-free pre-training from videos,","venue":null,"work_id":"18304667-5261-4cf3-afd0-1a0cc43d6747","year":2022},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.456781Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:0866d289032cb829ff4f2574ff3eb8ea69d9cb622770e894e4344f20b68be640","observation_id":"4208aab7-78ce-44ad-9058-3c828de85af9","resolution":{"observed_at":"2026-08-11T15:42:34.073417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.046972Z","title":"Video pretraining (vpt): Learning to act by watching unlabeled online videos,","venue":null,"work_id":"b7d81ba0-7f9b-4cc4-b6b6-876a604a9875","year":2022},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.461587Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:555a3172bdaa377b6a59c1710f38403ec6322e3c33ab772205d54eccbd9368a2","observation_id":"b65159b4-c32f-4350-9f56-7c3786898076","resolution":{"observed_at":"2026-08-11T15:42:34.053724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.466888Z","title":"Masked world models for visual control,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.466888Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:1afd6b1859441eeaafef52459312a37631a1d4ceefefd131127a42c3c5afcb1c","observation_id":"3fd6e46d-52bf-420a-9366-da5312ffe704","resolution":{"observed_at":"2026-08-11T15:42:33.466888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:34.015861Z","title":"Multi-view masked world models for visual robotic manipulation,","venue":null,"work_id":"474399cf-6c20-45ca-9997-49227356b406","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.472266Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:c43de9a9dbc35aefbeb7b208e01e29e82a6cab544049f57ae937ef7c02657145","observation_id":"4455f16d-f677-4089-a0dc-94fa816436f1","resolution":{"observed_at":"2026-08-11T15:42:34.022166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.993714Z","title":"Visual imitation learning with patch rewards,","venue":null,"work_id":"f7e8dc26-1c3d-443b-81e5-7fa548afdb7e","year":2023},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.477600Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:ef027981581f587a18ccbbcae4bacf03607852c34a26ea60bff38d99067f4a99","observation_id":"fbb167bd-cebe-4eae-97ac-d4e9651ad1e5","resolution":{"observed_at":"2026-08-11T15:42:34.000764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.970423Z","title":"Adversarial imitation learning from visual observations using latent information,","venue":null,"work_id":"526aa25b-81ad-41ee-9735-a34c61c532b6","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.483782Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:de65999b2ac9842935081882d7bcfa899d0fb5d0844a4e5f2a609ea6652d1ff6","observation_id":"3aded90c-8958-436e-88ab-58d609be43a6","resolution":{"observed_at":"2026-08-11T15:42:33.978096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.949679Z","title":"Zero-shot visual imitation,","venue":null,"work_id":"5b6f921c-0d7e-4cec-bd57-c8b52ad59f50","year":2018},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.488727Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:01c853da0bd67c5f001e519c0b2450f54d4c65b7eb8bdec246a1591edd43aeef","observation_id":"0cf2abea-33e2-44db-ba82-56717b014fbf","resolution":{"observed_at":"2026-08-11T15:42:33.956052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.493603Z","title":"Momentum contrast for unsupervised visual representation learning,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.493603Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:8958edcca3cc4940f655378dba06fdf7dfaaa2a03cc0eb524af01edf223aa911","observation_id":"aaad3a52-59a3-4268-8275-abd662dc58f6","resolution":{"observed_at":"2026-08-11T15:42:33.493603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.914572Z","title":"Decoupling rep- resentation learning from reinforcement learning,","venue":null,"work_id":"d969668f-21c6-4113-ba43-5f7f3a9d27f1","year":2021},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.499464Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:ac0f9d206073ed0a7ece1527b3aad1c00bcc25c94c5e7b199beba278448b6204","observation_id":"bfeb7c60-bc68-4402-9cd8-94a2f6da509e","resolution":{"observed_at":"2026-08-11T15:42:33.920941Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.895182Z","title":"A simple frame- work for contrastive learning of visual representations,","venue":null,"work_id":"91d4ae5e-56d8-47d4-8cb1-de7580ec2b3f","year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.505537Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:1e03fe1da7dafd7c88820e2ebb23e872b93d8e6c98eb7df8eb4d26f3975364c2","observation_id":"feb31420-7e35-44d4-8d7a-ec93ac6b9f5d","resolution":{"observed_at":"2026-08-11T15:42:33.901313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.870967Z","title":"On mutual information maximization for representation learning,","venue":null,"work_id":"1d3b4b6b-3a5f-4779-8674-1d1dbbad5f46","year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.511798Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:b597e3ebb1127ce57df98d564c63cb64c32173246c10a96d75b1f66d577a62b6","observation_id":"12841362-dd65-4661-95be-6252a9f9d839","resolution":{"observed_at":"2026-08-11T15:42:33.879892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.842633Z","title":"Neural discrete representa- tion learning,","venue":null,"work_id":"bcc47821-dc98-4f86-bde0-b214f46875a7","year":2017},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.517798Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:2afd03b2e8f578564717eb81d7b2a7b7845dea3c615a973dd98faa0acd6bf3d0","observation_id":"e1b9dc4f-4eb0-4fc2-bc0a-3b4554891fe1","resolution":{"observed_at":"2026-08-11T15:42:33.849930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.813784Z","title":"Learning to act without actions,","venue":null,"work_id":"f9d1b33a-aae3-4ddb-8446-eb2b33a321d0","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.523928Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:97893545b0bf60544ba50a3e03f9534f55e59e181e7ce4f7b39b08ceff129cf8","observation_id":"3d819b2d-8b6e-476b-a2de-763ed1cf4834","resolution":{"observed_at":"2026-08-11T15:42:33.819956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-08-20T07:04:06.309989Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-11T15:42:33.528953Z","title":"Proximal policy optimization algorithms,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.528953Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:3280c75c423202f9d1c86cf9e1fdf44ce1b057613b0ab5e3ff6eb7a17ea307bc","observation_id":"51af3637-8af6-41d7-90d3-d131e68ec2db","resolution":{"observed_at":"2026-08-11T15:42:33.528953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.788109Z","title":"Policy gradient without boostrapping via truncated value learning,","venue":null,"work_id":"061d9029-6d0e-45ee-97d9-4c32f5f51eb4","year":2024},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.534314Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:7e8255cddb1ab515b4d4900fe2a017e7f3e2ea42989df1909e44369b22a6d5a9","observation_id":"59126c88-becc-4a92-b06c-98e23ad77629","resolution":{"observed_at":"2026-08-11T15:42:33.795563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.759831Z","title":"Leveraging procedu- ral generation to benchmark reinforcement learning,","venue":null,"work_id":"119a2ff0-06c1-4f36-8b57-2f5ff3dd603e","year":2020},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.539412Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:afaed9e697b9b33189d0cf56e352154a0985f56d4a7425cf6a8f5789d0a4d6ec","observation_id":"4c2b921c-6c76-4ab1-8575-b56f270a0911","resolution":{"observed_at":"2026-08-11T15:42:33.767644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.736203Z","title":"Impala: Scalable dis- tributed deep-rl with importance weighted actor-learner architectures,","venue":null,"work_id":"19bb451f-da9c-46ce-a362-061b4bb29203","year":2018},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.547687Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:9dff7812a8f88c30e0383b8181d64c3eb56816eb98539f55ed48be5980b769fc","observation_id":"234d9e55-002e-4f43-85e9-5a0c83215da6","resolution":{"observed_at":"2026-08-11T15:42:33.745728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T15:42:33.553005Z","title":"U-net: Convolutional networks for biomedical image segmentation,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.553005Z"},"links":{"citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:ad4f70b93e24f2d613cb059e96e7654ce67c7238b6da618384307af61c8fb5b8","observation_id":"14cdab2b-6972-4560-a26f-975e8a12bdf4","resolution":{"observed_at":"2026-08-11T15:42:33.553005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-08-17T19:26:44.032537Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-11T15:42:33.558286Z","title":"Adam: A method for stochastic optimiza- tion,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-11T15:42:33.558286Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2412.10778"},"observation_digest":"sha256:ddc2c6be60837605fcdd5ebe7a10d4f4b9591c9a4c4f6ebeced09318e86505eb","observation_id":"bc1898d2-b97c-4428-8317-491bc9b8b80d","resolution":{"observed_at":"2026-08-11T15:42:33.558286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.10778","last_updated":"2025-04-08T08:54:33Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-18T08:58:15.616062Z","submitted_at":"2024-12-14T10:12:22Z","title":"Sample-efficient Unsupervised Policy Cloning from Ensemble Self-supervised Labeled Videos"},"reference_resolution":{"displayed":61,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":45},"total_outbound_references":61},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 61 of 61 outbound references and 0 inbound Pith citation observations for arXiv:2412.10778."}