{"as_of":"2026-08-13T15:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f5d6b29a496803e24b3e8dbaf0ec4e08e6369f4516cf12450dd40c471c043ae7","coverage":[{"denominator":105,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T21:07:40.137416Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T13:13:42.052386Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-10T13:20:26.104422Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"cited_work":{"arxiv_id":"2411.09145","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.09145","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-supervised monocular 4d scene reconstruction for egocentric videos","venue":null,"work_id":"c0771af3-406e-4579-811c-8941f21e5eff","year":2024},"citing_paper":{"arxiv_id":"2604.14025","last_updated":"2026-04-15T16:07:18Z","snapshot_observed_at":"2026-07-06T23:01:55.698452Z","submitted_at":"2026-04-15T16:07:18Z","title":"Feed-Forward 3D Scene Modeling: A Problem-Driven Perspective","version":1},"reference_index":188,"source":"pdf_text","source_observed_at":"2026-05-10T13:13:42.052386Z"},"links":{"cited_paper":"/paper/2411.09145","citing_paper":"/paper/2604.14025"},"observation_digest":"sha256:71877ad152ab9ad00db3a900629e565648331b617100807dafd338c97ee065d2","observation_id":"ed3defec-a7ea-4e41-a872-0303ba81f2a5","resolution":{"observed_at":"2026-05-10T13:20:26.107182Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.09145/citation-record","integrity":"/paper/2411.09145/integrity","json":"/paper/2411.09145/citation-record.json","paper":"/paper/2411.09145"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.722354Z","title":"Map-free visual relocalization: Metric pose relative to a single im- age","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.722354Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:09e1f986ebd65060f9038c4342c8ac1f808621aafb3c4103d83bc77ec9826a61","observation_id":"fc871b51-c499-4303-bab3-dacf8162cb8a","resolution":{"observed_at":"2026-08-12T21:07:39.722354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.727486Z","title":"Affordances from human videos as a versatile representation for robotics","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.727486Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:33c0bf7b6f31c32b214cf51e69b3c00dc21ad4010fee4022db63c0163ad771ee","observation_id":"26c7e916-6c5c-4731-957b-d3c2f3e253a5","resolution":{"observed_at":"2026-08-12T21:07:39.727486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.732033Z","title":"Uncertainty-aware state space transformer for egocentric 3d hand trajectory forecasting","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.732033Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:64336a50a21f9dfad117a0efd9aaf7decfc62943f5dc2138b67522e87c43c15b","observation_id":"62af5fcf-ab0f-49d1-a46f-fd0ce6277d14","resolution":{"observed_at":"2026-08-12T21:07:39.732033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.12288","last_updated":"2023-02-23T19:13:10Z","snapshot_observed_at":"2026-07-06T14:55:15.719380Z","submitted_at":"2023-02-23T19:13:10Z","title":"ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.12288","snapshot_observed_at":"2026-08-12T21:07:39.736384Z","title":"Zoedepth: Zero-shot trans- fer by combining relative and metric depth","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.736384Z"},"links":{"cited_paper":"/paper/2302.12288","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:de96ba8c900d48c126e5000d62f81a1ccfe06d36786a1068a12878e522d2c31c","observation_id":"4cc3440b-6392-450e-9c64-c491634de4f5","resolution":{"observed_at":"2026-08-12T21:07:39.736384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.740804Z","title":"Unsuper- vised scale-consistent depth and ego-motion learning from monocular video","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.740804Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:68003418f296a4c9b9a52fdf7573d7ff75208e0f64b8f84d38bc27bcedce294a","observation_id":"9065cc68-ab5f-4f63-8b2a-e69e1d536228","resolution":{"observed_at":"2026-08-12T21:07:39.740804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.14460","last_updated":"2023-07-26T19:01:49Z","snapshot_observed_at":"2026-08-13T10:47:18.882919Z","submitted_at":"2023-07-26T19:01:49Z","title":"MiDaS v3.1 -- A Model Zoo for Robust Monocular Relative Depth Estimation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.14460","snapshot_observed_at":"2026-08-12T21:07:39.745048Z","title":"Midas v3","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.745048Z"},"links":{"cited_paper":"/paper/2307.14460","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:d73dedc390532073305a216e31660f3838a987ea1891d7fdda782fa96f4975db","observation_id":"833882f7-bd0c-4c9c-9582-8167e6ca95a9","resolution":{"observed_at":"2026-08-12T21:07:39.745048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.749526Z","title":"Orb-slam3: An ac- curate open-source library for visual, visual–inertial, and multimap slam","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.749526Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:4772f0915abda9bda9ac22dba6ec745788e80a916debe0ea363c3e96396d6f5b","observation_id":"1bad3f8f-7ece-4ea4-b145-a79d8cb0839d","resolution":{"observed_at":"2026-08-12T21:07:39.749526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.753033Z","title":"Leap-vo: Long-term effective any point tracking for visual odometry","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.753033Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:2d87cf5fe1adbca30fe2fc6f0e83b56d7a8a114d789a1a0d047966c1d65a59b0","observation_id":"a1b3097b-edce-47ed-8e39-7db7f4ac045e","resolution":{"observed_at":"2026-08-12T21:07:39.753033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.08534","last_updated":"2023-02-13T15:50:22Z","snapshot_observed_at":"2026-08-13T15:42:29.664753Z","submitted_at":"2022-05-17T17:59:11Z","title":"Vision Transformer Adapter for Dense Predictions","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.08534","snapshot_observed_at":"2026-08-12T21:07:39.756515Z","title":"Vision transformer adapter for dense predictions","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.756515Z"},"links":{"cited_paper":"/paper/2205.08534","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:7a42fb363f0d2a097ed61875a09abe8c8aa9ce08c66c0ab7cea169e4bab94854","observation_id":"9407b8ef-acd2-40a5-9e72-155f7d6edcf2","resolution":{"observed_at":"2026-08-12T21:07:39.756515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.761226Z","title":"Treating mo- tion as option to reduce motion dependency in unsuper- vised video object segmentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.761226Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:4ad1fd57cd569ee6d9775040dc4a44e2efdf1f3df862bc9a415a2eeb720c4508","observation_id":"0bdcb2d6-179f-4774-8467-20c23a8c49d7","resolution":{"observed_at":"2026-08-12T21:07:39.761226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.765117Z","title":"Deep global registration","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.765117Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:c4e7fc73bd5b0ffd3a1c147e393e1fd5dbb359097734f9d76f76f900e0392de6","observation_id":"b5c51ec0-2d47-4095-b1e3-1b9dd05a276e","resolution":{"observed_at":"2026-08-12T21:07:39.765117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02280","last_updated":"2024-05-23T16:27:11Z","snapshot_observed_at":"2026-08-13T00:15:12.213891Z","submitted_at":"2024-05-03T17:55:34Z","title":"DreamScene4D: Dynamic Multi-Object Scene Generation from Monocular Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.02280","snapshot_observed_at":"2026-08-12T21:07:39.769174Z","title":"Dreamscene4d: Dynamic multi-object scene generation from monocular videos","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.769174Z"},"links":{"cited_paper":"/paper/2405.02280","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:faca3e45469d20f5e540a9693c8f805635bbf2c5ba379fa7f292b4987bfc7421","observation_id":"4f4aaee8-351a-4009-982b-7127eae75a6f","resolution":{"observed_at":"2026-08-12T21:07:39.769174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.773462Z","title":"3d u-net: learn- ing dense volumetric segmentation from sparse annota- tion","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.773462Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:e13ef49c03c7997922bfb5ee62154308c8c1233afce2530a993cc959b24eab83","observation_id":"e15d13c9-ea4b-4ef7-868c-8e97b7952971","resolution":{"observed_at":"2026-08-12T21:07:39.773462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.778342Z","title":"Global structure-from-motion by similarity averaging","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.778342Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:314c7b25cdd6c18993aa562c29ec12228766b3a0a9fd5b2374bf47c2d4544bb5","observation_id":"e5f0714b-1fc7-40da-a2af-1a2b722cf532","resolution":{"observed_at":"2026-08-12T21:07:39.778342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.782785Z","title":"Rescaling egocentric vision: Collection, pipeline and chal- lenges for epic-kitchens-100","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.782785Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:0a08048267914055aa348d324fbbb5251315f1f707c507d20274ee7f42bd95cf","observation_id":"f4c9634c-7571-4226-bbd5-e64b9624be02","resolution":{"observed_at":"2026-08-12T21:07:39.782785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-12T21:07:39.787070Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.787070Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:b43509db96434abbda2ee7cdf833c38e87c240f511b58866be05f96f82a277f0","observation_id":"ed8293ed-8ba4-4990-9f28-013ee4d55557","resolution":{"observed_at":"2026-08-12T21:07:39.787070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.19152","last_updated":"2024-09-27T21:29:58Z","snapshot_observed_at":"2026-08-12T22:35:43.319224Z","submitted_at":"2024-09-27T21:29:58Z","title":"MASt3R-SfM: a Fully-Integrated Solution for Unconstrained Structure-from-Motion","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.19152","snapshot_observed_at":"2026-08-12T21:07:39.792383Z","title":"Mast3r-sfm: a fully-integrated solution for unconstrained structure-from-motion","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.792383Z"},"links":{"cited_paper":"/paper/2409.19152","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:f621f542c07f09f05d04a6c211f4fa6a37282e784d9e388191c42331d91da341","observation_id":"8c0911ff-fafe-4898-9ab5-e5bee2ac9ded","resolution":{"observed_at":"2026-08-12T21:07:39.792383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.04382","last_updated":"2024-05-02T16:17:25Z","snapshot_observed_at":"2026-08-13T15:47:39.730834Z","submitted_at":"2022-05-09T15:35:33Z","title":"FlowBot3D: Learning 3D Articulation Flow to Manipulate Articulated Objects","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.04382","snapshot_observed_at":"2026-08-12T21:07:39.797146Z","title":"Flowbot3d: Learning 3d articulation flow to manipulate articulated ob- jects","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.797146Z"},"links":{"cited_paper":"/paper/2205.04382","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:78d4187b04f526ea112a9d23b366a4c038fbf34d824257ef4bc41545f4ce9fff","observation_id":"c3d7c593-8126-46ac-bf5a-19ab458f272f","resolution":{"observed_at":"2026-08-12T21:07:39.797146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14914","last_updated":"2025-01-24T20:46:04Z","snapshot_observed_at":"2026-08-12T02:46:24.021753Z","submitted_at":"2025-01-24T20:46:04Z","title":"Light3R-SfM: Towards Feed-forward Structure-from-Motion","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14914","snapshot_observed_at":"2026-08-12T21:07:39.801035Z","title":"Light3r-sfm: Towards feed-forward structure- from-motion","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.801035Z"},"links":{"cited_paper":"/paper/2501.14914","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:2c1a797b2c51ab4590e15620313b2833bba1fc8acb06baf7c1d25ac841d564ef","observation_id":"3292ec59-0f45-4b18-9a05-210ee33f3a1b","resolution":{"observed_at":"2026-08-12T21:07:39.801035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.805305Z","title":"Arctic: A dataset for dexterous bimanual hand- object manipulation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.805305Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:2a4189467189a8c851e72698bcc0ba99912f4f1227087e5be8d0ddcf5f4866d8","observation_id":"35be13ad-6993-474c-9e69-42332e07921e","resolution":{"observed_at":"2026-08-12T21:07:39.805305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.809042Z","title":"Kaolin: A pytorch library for accelerating 3d deep learning research","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.809042Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:90d9655757a76c705f34c2914601ec2cca2d3cd0ff458fd6b0bcf1396f3c82b6","observation_id":"ae25698e-2d92-4cd6-9f0b-0dd4687260c8","resolution":{"observed_at":"2026-08-12T21:07:39.809042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.813130Z","title":"First-person hand action bench- mark with rgb-d videos and 3d hand pose annotations","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.813130Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:0db62dec2c5cf878be3eb73a4c9deaba745e6ae02369b289a37091b9eb8c618c","observation_id":"1c7d14fc-90b6-45a4-9c5e-40ecf0138573","resolution":{"observed_at":"2026-08-12T21:07:39.813130Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.816847Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.816847Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:30871da95439974b160daaa4a368635bf36ad805ab88c94021ed4ea23bfe9757","observation_id":"b3cc91b5-5adf-4e96-9c31-f6075fc1ac7c","resolution":{"observed_at":"2026-08-12T21:07:39.816847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.820503Z","title":"Deep relu networks have surprisingly few activation patterns","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.820503Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:91c2abf5417bc6815e2dea7bf5564ee9b1b571a87eae356ca4a30deb3f40382d","observation_id":"019b77d5-f088-438a-a5e7-c825514ff200","resolution":{"observed_at":"2026-08-12T21:07:39.820503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02095","last_updated":"2024-11-27T07:59:25Z","snapshot_observed_at":"2026-08-12T22:52:07.327467Z","submitted_at":"2024-09-03T17:52:03Z","title":"DepthCrafter: Generating Consistent Long Depth Sequences for Open-world Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02095","snapshot_observed_at":"2026-08-12T21:07:39.824475Z","title":"Depthcrafter: Generating consistent long depth sequences for open-world videos","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.824475Z"},"links":{"cited_paper":"/paper/2409.02095","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:7a65a5c95affd54e6de7c34c49f63b1cdea859c623ed8cbe5b278cc883e164dd","observation_id":"2b4b57a4-5191-499a-ac66-8d0c8262d8ca","resolution":{"observed_at":"2026-08-12T21:07:39.824475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.09621","last_updated":"2025-04-30T17:59:59Z","snapshot_observed_at":"2026-08-12T19:47:06.298035Z","submitted_at":"2024-12-12T18:59:54Z","title":"Stereo4D: Learning How Things Move in 3D from Internet Stereo Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.09621","snapshot_observed_at":"2026-08-12T21:07:39.829236Z","title":"Stereo4d: Learning how things move in 3d from internet stereo videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.829236Z"},"links":{"cited_paper":"/paper/2412.09621","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:d13d7028e8400d35722d6941dde34fdf6139f0c4ad58c9cd844c9b804337cb29","observation_id":"13a25523-1650-48aa-9a76-c70de8e6d72c","resolution":{"observed_at":"2026-08-12T21:07:39.829236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.07635","last_updated":"2024-10-01T13:15:53Z","snapshot_observed_at":"2026-08-13T10:55:12.508649Z","submitted_at":"2023-07-14T21:13:04Z","title":"CoTracker: It is Better to Track Together","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.07635","snapshot_observed_at":"2026-08-12T21:07:39.833351Z","title":"Co- tracker: It is better to track together","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.833351Z"},"links":{"cited_paper":"/paper/2307.07635","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:d6b5dc41d21ec5dc136ec38cfa1f21a6fddca9e19e228c8ddfcc433ac72c4556","observation_id":"fa2ebabf-c41f-448f-82fe-c1b11a3272c3","resolution":{"observed_at":"2026-08-12T21:07:39.833351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.837229Z","title":"Fast encoder- based 3d from casual videos via point track processing","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.837229Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:a7fd7a3dfa4f9599b42499d22e1006147f698eacf5d5016bee4f804c690830d7","observation_id":"0cad78d6-a241-470d-ac6b-00b0a4c4eb52","resolution":{"observed_at":"2026-08-12T21:07:39.837229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.841062Z","title":"3d gaussian splatting for real-time radiance field rendering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.841062Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:878b1885b71e8f5dcd60077f8aac92e00a124c910872ffe61f94d1f428dffb8d","observation_id":"fa4ded9c-035d-4aff-8d63-78e54952c664","resolution":{"observed_at":"2026-08-12T21:07:39.841062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-12T21:07:39.845079Z","title":"Adam: A method for stochastic opti- mization","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.845079Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:90176480db4139514a97b63b572eeaee561612ab48b5827514ec28f5537f0e4d","observation_id":"d62c97b7-dcb3-4e44-b01e-b24637a62dd4","resolution":{"observed_at":"2026-08-12T21:07:39.845079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.849213Z","title":"Segment anything","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.849213Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:8339ec231f5eb6eaa1c11e4afca23009deac27b7f50422149414fce8abf498b6","observation_id":"7d0781c7-f5fc-4083-b656-ae365e645e0f","resolution":{"observed_at":"2026-08-12T21:07:39.849213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.852914Z","title":"Ro- bust consistent video depth estimation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.852914Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:78fadbc13fd27cf2c741e6cfa839c73a03e255b8c6e7c7bae7e69b73e166e17b","observation_id":"ad4f867f-7741-4622-a6c9-c46622d340f3","resolution":{"observed_at":"2026-08-12T21:07:39.852914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05921","last_updated":"2024-08-27T17:14:16Z","snapshot_observed_at":"2026-08-13T15:32:31.588641Z","submitted_at":"2024-07-08T13:28:47Z","title":"TAPVid-3D: A Benchmark for Tracking Any Point in 3D","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05921","snapshot_observed_at":"2026-08-12T21:07:39.856645Z","title":"Tapvid-3d: A benchmark for tracking any point in 3d","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.856645Z"},"links":{"cited_paper":"/paper/2407.05921","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:f4da45128119937a006ff582f6229a79ed447ced497016550e3e4b12214b996c","observation_id":"e267ad21-5df7-4496-b9e5-875670612683","resolution":{"observed_at":"2026-08-12T21:07:39.856645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.861008Z","title":"H2o: Two hands manipulating objects for first person interaction recognition","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.861008Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:bd8917423e6c8b94a81f8dd0159eb658e3bd6f1b7a3374aba0dbaa2ff460715e","observation_id":"a581af7c-ea2a-49d6-b74e-c2aeb18a21e4","resolution":{"observed_at":"2026-08-12T21:07:39.861008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17421","last_updated":"2024-11-29T18:53:12Z","snapshot_observed_at":"2026-08-12T23:56:50.961999Z","submitted_at":"2024-05-27T17:59:07Z","title":"MoSca: Dynamic Gaussian Fusion from Casual Videos via 4D Motion Scaffolds","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.17421","snapshot_observed_at":"2026-08-12T21:07:39.864731Z","title":"Mosca: Dynamic gaussian fusion from casual videos via 4d motion scaffolds","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.864731Z"},"links":{"cited_paper":"/paper/2405.17421","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:41756bcb9b20e07450db67200a2c0c5da29784bb079ce0e3ab2f730d6c915be0","observation_id":"771f7275-ead0-4751-930e-73dcd92293d1","resolution":{"observed_at":"2026-08-12T21:07:39.864731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.09756","last_updated":"2024-06-14T06:46:30Z","snapshot_observed_at":"2026-08-12T23:42:59.235727Z","submitted_at":"2024-06-14T06:46:30Z","title":"Grounding Image Matching in 3D with MASt3R","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.09756","snapshot_observed_at":"2026-08-12T21:07:39.869186Z","title":"Grounding image matching in 3d with mast3r","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.869186Z"},"links":{"cited_paper":"/paper/2406.09756","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:8c2838c682bee306ba72d1b23a7be2426cf3a24b5cfcd541393079bcfc068404","observation_id":"410c6c70-28ca-4277-9c74-ad7d54ac8563","resolution":{"observed_at":"2026-08-12T21:07:39.869186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.873375Z","title":"Egocentric pre- diction of action target in 3d","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.873375Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:a41ac73ac47e9a7f7c27176627eb4b09db35906e5e287b5296a1d1cac4d20b46","observation_id":"6bbdfc7a-5bd7-4253-934f-717519d1ca98","resolution":{"observed_at":"2026-08-12T21:07:39.873375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04463","last_updated":"2024-12-06T19:15:46Z","snapshot_observed_at":"2026-08-11T21:21:34.426003Z","submitted_at":"2024-12-05T18:59:42Z","title":"MegaSaM: Accurate, Fast, and Robust Structure and Motion from Casual Dynamic Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04463","snapshot_observed_at":"2026-08-12T21:07:39.876857Z","title":"Megasam: Accurate, fast, and robust structure and motion from casual dynamic videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.876857Z"},"links":{"cited_paper":"/paper/2412.04463","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:a947a12322acc019db5a862656390b93d15d1fbb9b8eb68f8d9743d5efbfb42f","observation_id":"360c94b3-c92c-473a-bf4c-538f9bf39895","resolution":{"observed_at":"2026-08-12T21:07:39.876857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.881571Z","title":"Feed-forward bullet-time reconstruction of dynamic scenes from monoc- ular videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.881571Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:75aa77a079de6cd10f64b3db607952a7ce065412a6aebad7e462ae951b56de8d","observation_id":"fb0ff72c-6d51-4d1e-a093-473f27697463","resolution":{"observed_at":"2026-08-12T21:07:39.881571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.886105Z","title":"Few- shot parameter-efficient fine-tuning is better and cheaper than in-context learning","venue":null,"work_id":null,"year":1950},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.886105Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:a324637f981b98805060bc6d693381c8996e410b7ec1716f09c60913edc115af","observation_id":"7e37d1ab-36be-4a71-aa3e-5b105c8587d5","resolution":{"observed_at":"2026-08-12T21:07:39.886105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-08-12T21:07:39.890781Z","title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.890781Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:61b9f630267302e8641e7826770b407179a26919f5294c2f86b90deacf103074","observation_id":"ba609dcf-acb6-4e2b-9b98-02be8df5d7ab","resolution":{"observed_at":"2026-08-12T21:07:39.890781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.895402Z","title":"Joint esti- mation of pose, depth, and optical flow with a competition– cooperation transformer network","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.895402Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:ba86a2c3ebe933e23befb7cc0f1affdd6616cd60c2957264b8243cd0508f3aeb","observation_id":"1dc9a8f6-7f05-422c-8669-29f0d104494c","resolution":{"observed_at":"2026-08-12T21:07:39.895402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.899433Z","title":"Hoi4d: A 4d egocentric dataset for category-level human-object interaction","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.899433Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:c7bc1ba81af09ec30dcef9b492ab0db6d2e689c3f36b28ea0dfa48e631c18776","observation_id":"5cf17ea6-3ca9-440a-9034-3e63a67ba67b","resolution":{"observed_at":"2026-08-12T21:07:39.899433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.903113Z","title":"Robust dynamic radi- ance fields","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.903113Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:fe52a2be273ea06fa2f1483ef22d85ed5f7823c2c28c219711ca752aa8395d20","observation_id":"01478555-27c0-44ca-88c7-39f22bf8443b","resolution":{"observed_at":"2026-08-12T21:07:39.903113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.03079","last_updated":"2024-12-05T14:16:07Z","snapshot_observed_at":"2026-08-12T06:04:56.696888Z","submitted_at":"2024-12-04T07:09:59Z","title":"Align3R: Aligned Monocular Depth Estimation for Dynamic Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.03079","snapshot_observed_at":"2026-08-12T21:07:39.907353Z","title":"Align3r: Aligned monocu- lar depth estimation for dynamic videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.907353Z"},"links":{"cited_paper":"/paper/2412.03079","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:2fad8fdf6f22cd1ae6ca8b1d75e4f3b2f4d29926c1ddecf8375cce954a78a750","observation_id":"fd752f31-870b-4b56-b35c-98f93301cdfa","resolution":{"observed_at":"2026-08-12T21:07:39.907353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.09713","last_updated":"2023-08-18T17:59:21Z","snapshot_observed_at":"2026-08-13T10:32:10.871509Z","submitted_at":"2023-08-18T17:59:21Z","title":"Dynamic 3D Gaussians: Tracking by Persistent Dynamic View Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.09713","snapshot_observed_at":"2026-08-12T21:07:39.911685Z","title":"Dynamic 3d gaussians: Tracking by persistent dynamic view synthesis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.911685Z"},"links":{"cited_paper":"/paper/2308.09713","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:285828895bfb802b65fc2c368a268ab5072238770bc65f095a788cdb2c3b3fe7","observation_id":"dfd32bd4-5a49-416a-9492-7b11a166636f","resolution":{"observed_at":"2026-08-12T21:07:39.911685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.115293Z","title":"Object scene flow for autonomous vehicles","venue":null,"work_id":"b43db776-20b6-4f82-a884-765513074156","year":2015},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.916040Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:c6122463af6d3d1009c17a4b601e82fa34a59bf822ab0e51e10c7584cf3fef04","observation_id":"d117891a-ff1d-4d7a-802a-02b937b76f84","resolution":{"observed_at":"2026-08-12T21:07:41.119232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.103871Z","title":"Embodiedgpt: Vision-language pre-training via embodied chain of thought","venue":null,"work_id":"c02a4bad-b604-40d2-b734-20e59e3768d7","year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.919907Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:e47b7a29af2da53eb215cd88ffef8c0d50d3cfb70471b81d5bd7b9c5e463b7fe","observation_id":"4f9548d2-61f9-4a74-9ce5-ae3b0080225b","resolution":{"observed_at":"2026-08-12T21:07:41.107864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.924466Z","title":"Orb-slam: a versatile and accurate monocular slam system","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.924466Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:30f6f3509304911e33d9ef6b7270ab42685ea94d6c4b3b4d3b518ff88b847b88","observation_id":"bd638edc-3ab8-44e7-921e-f00c6555a8cf","resolution":{"observed_at":"2026-08-12T21:07:39.924466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12392","last_updated":"2025-06-02T13:44:10Z","snapshot_observed_at":"2026-08-12T21:07:35.404547Z","submitted_at":"2024-12-16T23:00:05Z","title":"MASt3R-SLAM: Real-Time Dense SLAM with 3D Reconstruction Priors","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12392","snapshot_observed_at":"2026-08-12T21:07:39.928959Z","title":"Mast3r-slam: Real-time dense slam with 3d reconstruction priors","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.928959Z"},"links":{"cited_paper":"/paper/2412.12392","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:773825d767945d277b0f095cf098a8416409f192809f884046216ef20d920479","observation_id":"90def821-ff91-4fc9-a4d2-b9668621b832","resolution":{"observed_at":"2026-08-12T21:07:39.928959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.12601","last_updated":"2022-11-18T05:57:09Z","snapshot_observed_at":"2026-08-11T08:36:53.636356Z","submitted_at":"2022-03-23T17:55:09Z","title":"R3M: A Universal Visual Representation for Robot Manipulation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.12601","snapshot_observed_at":"2026-08-12T21:07:39.932997Z","title":"R3m: A universal visual representation for robot manipulation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.932997Z"},"links":{"cited_paper":"/paper/2203.12601","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:7d53331e433553a5b0bb33a331284419fb903d13ec955a59c0b6130a17ee0191","observation_id":"a3ffc945-c7e4-4239-baa0-232cc9c90bd8","resolution":{"observed_at":"2026-08-12T21:07:39.932997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07193","last_updated":"2024-02-02T10:24:09Z","snapshot_observed_at":"2026-08-11T10:12:11.384939Z","submitted_at":"2023-04-14T15:12:19Z","title":"DINOv2: Learning Robust Visual Features without Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07193","snapshot_observed_at":"2026-08-12T21:07:39.937434Z","title":"Dinov2: Learning robust visual features without supervi- sion","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.937434Z"},"links":{"cited_paper":"/paper/2304.07193","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:639324c0466523119f1101a553dce8b799bc1b7031f040a1181c778d551ce5b5","observation_id":"37649b4a-a0dd-49c2-95d0-313d21ccfe1c","resolution":{"observed_at":"2026-08-12T21:07:39.937434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.08122","last_updated":"2022-03-15T17:50:54Z","snapshot_observed_at":"2026-08-05T03:04:12.284957Z","submitted_at":"2022-03-15T17:50:54Z","title":"From 2D to 3D: Re-thinking Benchmarking of Monocular Depth Prediction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.08122","snapshot_observed_at":"2026-08-12T21:07:39.942006Z","title":"From 2d to 3d: Re-thinking benchmarking of monocular depth predic- tion","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.942006Z"},"links":{"cited_paper":"/paper/2203.08122","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:863ce50653ece10cc782af7fd07e0ff8e37f848a6ff5fa76bfaa7f73b75cce35","observation_id":"c0ee9492-3e9d-4978-9050-c15b7d10b3f7","resolution":{"observed_at":"2026-08-12T21:07:39.942006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.084868Z","title":"Aria digital twin: A new benchmark dataset for egocentric 3d machine perception","venue":null,"work_id":"2a102e01-9df9-4ed4-bc1d-104cae940293","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.946552Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:43a728853d0709b058abccb5fd7d76f0b7b957b227d82494bc517185cd4160c1","observation_id":"a9ae9f06-fcad-4853-8292-49a4ced0968f","resolution":{"observed_at":"2026-08-12T21:07:41.088733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.073810Z","title":"Re- constructing hands in 3d with transformers","venue":null,"work_id":"45d7996e-5219-4d85-ae81-a2dabeaa52c1","year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.951025Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:f3291285eb7f6291f8213660efd19cec31bbf442f29de74b12dd738c6ee1f32c","observation_id":"c5a6e2b5-5371-4552-99c2-617ae639307a","resolution":{"observed_at":"2026-08-12T21:07:41.077778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.062423Z","title":"Unidepth: Universal monocular metric depth estimation","venue":null,"work_id":"667313b4-3ac9-4247-af41-40f15fd35b1f","year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.955421Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:d5229c7591c325f18ef76102a0391474e543eb08bbf49e297a0b5665f22ed348","observation_id":"d49c8d1c-1fcd-4133-9c2c-9130058fbc12","resolution":{"observed_at":"2026-08-12T21:07:41.066349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12259","last_updated":"2025-03-26T18:05:52Z","snapshot_observed_at":"2026-08-13T01:07:29.541790Z","submitted_at":"2024-09-18T18:46:51Z","title":"WiLoR: End-to-end 3D Hand Localization and Reconstruction in-the-wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12259","snapshot_observed_at":"2026-08-12T21:07:39.959374Z","title":"Wilor: End-to-end 3d hand localization and reconstruction in-the-wild","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.959374Z"},"links":{"cited_paper":"/paper/2409.12259","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:769831fe8de34c9c9fd889029b50495060d1be9458d00459542b7426447e2907","observation_id":"56050f13-69cd-4ce0-aaa0-2c134170a9a0","resolution":{"observed_at":"2026-08-12T21:07:39.959374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.049855Z","title":"Egovlpv2: Egocentric video-language pre-training with fusion in the backbone","venue":null,"work_id":"eaee5961-5c28-4874-bc3a-eaf1bf03c794","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.964439Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:ea56656b5aedaaaa94933be8c488fcef66c1512bee98fac12645dd67352fcc3c","observation_id":"6458bab8-fcf3-4ce5-87d4-ac5f6ef1f5c4","resolution":{"observed_at":"2026-08-12T21:07:41.054606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.968803Z","title":"Affordancellm: Grounding affordance from vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.968803Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:0b05ebd6407e204bcfecd08f8a2f8ac6233a7e2bf6acf9ab18d6414de9c2fa99","observation_id":"a123a616-77b1-4ee9-bf5e-00fe23221957","resolution":{"observed_at":"2026-08-12T21:07:39.968803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.029830Z","title":"From one hand to multiple hands: Imitation learning for dexterous manipula- tion from single-camera teleoperation","venue":null,"work_id":"62e391fd-6806-489a-9972-efeeee0fd67e","year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.973093Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:dcbb7411c374e104d38634e32f67a05ec270accdf3d7b045a779032dcfecf7f0","observation_id":"53e1322a-f990-46de-8941-22f71717d591","resolution":{"observed_at":"2026-08-12T21:07:41.034084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.017748Z","title":"Deep learning- based depth estimation methods from monocular image and videos: A comprehensive survey.ACM Computing Surveys,","venue":null,"work_id":"c0846e63-19c2-4d89-bdb7-56d6dd58c562","year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.976711Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:0149bd12d963293a4738511412385f1ad80328cb74459469d5438ff54c7f6671","observation_id":"6d3558cb-8b35-404e-8c73-0d423e7ed2f9","resolution":{"observed_at":"2026-08-12T21:07:41.021719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:41.006410Z","title":"Nerf- slam: Real-time dense monocular slam with neural radi- ance fields","venue":null,"work_id":"1358cb5d-31cf-4a69-9e44-a0c7fbf1da57","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.980151Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:54788f88c8c7541c0862b82dd9f458e76641cedea84ea3508edc29851b7daf1a","observation_id":"2b4e34d6-8d0b-4262-9a6c-9cd953a919bb","resolution":{"observed_at":"2026-08-12T21:07:41.010196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.994012Z","title":"Identification and cor- rection of flying pixels in range camera data","venue":null,"work_id":"ade41445-8eb1-4d24-8355-f5473307ce05","year":2008},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.983646Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:7290f2b00e963b58485fe3ae8f8f220abe91223c8f7ab0922aa84252e65fb808","observation_id":"64366714-3340-4ea4-9237-bfffea7ee635","resolution":{"observed_at":"2026-08-12T21:07:40.998289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.982762Z","title":"Structure-from-motion revisited","venue":null,"work_id":"c3dcf7e9-d7f7-41b0-8152-15c5cdea0d32","year":2016},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.986988Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:be4911d7543d47d30f33cb294999142666103b9d6b80e0f591ab310711a937ef","observation_id":"e8f264e2-97a4-4a00-9f6e-be3485b35b25","resolution":{"observed_at":"2026-08-12T21:07:40.986795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.990585Z","title":"Pixelwise view selection for unstructured multi-view stereo","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.990585Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:69da73d750405935e84017ef9118d0403fecd1588c9a03f09a225e83151f9624","observation_id":"261e3b0a-db13-4d3b-8861-33545c495bd4","resolution":{"observed_at":"2026-08-12T21:07:39.990585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.965362Z","title":"Toolflownet: Robotic manipulation with tools via predicting tool flow from point clouds","venue":null,"work_id":"05d20625-a507-490a-9781-da63b8dde1ed","year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.994388Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:4120293e0a77c2c217494958f72f804abfe67444893097552cca9c0108c3ba22","observation_id":"0146b440-b9c6-4b03-88d6-57e184c6ac07","resolution":{"observed_at":"2026-08-12T21:07:40.969211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:39.998249Z","title":"Understanding human hands in contact at inter- net scale","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:39.998249Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:c2aa2fbb347052855102edb7235a10ab2b94b1cd29d5f2d490e18bec877f28eb","observation_id":"32cea04f-46f4-4216-8269-8190c63af2b8","resolution":{"observed_at":"2026-08-12T21:07:39.998249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.946475Z","title":"Motion-based object segmentation based on dense rgb-d scene flow","venue":null,"work_id":"d905ef2b-556b-42d4-857b-fc8928b568a6","year":2018},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.002167Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:b5bd96d3c397b518e4116061d83227f755a0e52971719804f9b3594880075c7c","observation_id":"b18c3a36-56f6-4411-9d71-decce3f49de7","resolution":{"observed_at":"2026-08-12T21:07:40.950347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.933649Z","title":"Swindepth: Unsupervised depth estimation using monocular sequences via swin trans- former and densely cascaded network","venue":null,"work_id":"8a54ae7f-ba56-47c2-96ae-7cbb77ba0c10","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.005743Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:0cb589bef35cd79cc0b15d3b046bcf97b6f852332d8117217ad94bb967ade81c","observation_id":"d9096ce8-f580-4d19-8bfc-969a377b5f3a","resolution":{"observed_at":"2026-08-12T21:07:40.938047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00180","last_updated":"2023-05-31T20:58:46Z","snapshot_observed_at":"2026-08-13T11:27:54.156256Z","submitted_at":"2023-05-31T20:58:46Z","title":"FlowCam: Training Generalizable 3D Radiance Fields without Camera Poses via Pixel-Aligned Scene Flow","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00180","snapshot_observed_at":"2026-08-12T21:07:40.010002Z","title":"Flowcam: training generalizable 3d radiance fields without camera poses via pixel-aligned scene flow","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.010002Z"},"links":{"cited_paper":"/paper/2306.00180","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:89c5120316c644ec60631b8f3fd3dc99781aa295fab12e7f494ccee8519406b6","observation_id":"76066f33-1564-49f8-ac58-76d48c533607","resolution":{"observed_at":"2026-08-12T21:07:40.010002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15259","last_updated":"2024-07-23T13:41:03Z","snapshot_observed_at":"2026-08-13T00:23:36.061824Z","submitted_at":"2024-04-23T17:46:50Z","title":"FlowMap: High-Quality Camera Poses, Intrinsics, and Depth via Gradient Descent","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15259","snapshot_observed_at":"2026-08-12T21:07:40.014191Z","title":"Flowmap: High-quality camera poses, in- trinsics, and depth via gradient descent","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.014191Z"},"links":{"cited_paper":"/paper/2404.15259","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:0cd3553a6ee91486031a939a7958c6e8f100a1fc28c6d9d442aa0fb7f47d0f83","observation_id":"4e6e7d76-16df-41e5-973c-9dfc83ff6167","resolution":{"observed_at":"2026-08-12T21:07:40.014191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.921157Z","title":"Kick back & relax: Learning to reconstruct the world by watching slowtv","venue":null,"work_id":"539f4f0b-bb9e-4a7b-b50a-168be6352b02","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.018770Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:8832fddb2fe80ce019f4fd5338fbe9017dd26792f4827b112807d968528aa063","observation_id":"82a7edd6-6f8a-4efb-a9d2-cd7501766a42","resolution":{"observed_at":"2026-08-12T21:07:40.925633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.01569","last_updated":"2024-03-03T17:29:03Z","snapshot_observed_at":"2026-08-13T04:04:36.014543Z","submitted_at":"2024-03-03T17:29:03Z","title":"Kick Back & Relax++: Scaling Beyond Ground-Truth Depth with SlowTV & CribsTV","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.01569","snapshot_observed_at":"2026-08-12T21:07:40.023727Z","title":"Kick back & relax++: Scaling beyond ground- truth depth with slowtv & cribstv","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.023727Z"},"links":{"cited_paper":"/paper/2403.01569","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:d0e3362713a9d9f35a12e55cb6b202c6819c5ebde706ccd2ca396a970d29a0a3","observation_id":"fda441e7-d31b-4993-a376-bfe52e239b3c","resolution":{"observed_at":"2026-08-12T21:07:40.023727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.909176Z","title":"A unified transformer frame- work for group-based segmentation: Co-segmentation, co- saliency detection and video salient object detection","venue":null,"work_id":"edade580-7f02-473b-9c34-da2b4bdffaf5","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.028807Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:2976730f012e1037667bce0e312b96cb64fc79d62aed048b5ef1afda9780146d","observation_id":"7f551f1d-ad97-4523-99c2-1ccba4e3bd5c","resolution":{"observed_at":"2026-08-12T21:07:40.913083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.032635Z","title":"3dgstream: On-the-fly training of 3d gaussians for efficient streaming of photo-realistic free- viewpoint videos","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.032635Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:6e9e556483e5ed891765e74b4734301b7671c001dd4181965d1557ac6ec8557c","observation_id":"3e1ea8e8-1847-4d64-8e09-16be8973f8ae","resolution":{"observed_at":"2026-08-12T21:07:40.032635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.890636Z","title":"Dynamo-depth: fix- ing unsupervised depth estimation for dynamical scenes","venue":null,"work_id":"bea89e9f-9a83-4e1d-9ed1-4ebd04409c31","year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.036984Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:4515030e95b4b1483d48427e381e8f7a69d468955e4040d4b040bcbc6f896cf0","observation_id":"9a30a8a0-d893-495f-b6a3-bfa0c3f21d25","resolution":{"observed_at":"2026-08-12T21:07:40.894609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.040954Z","title":"Droid-slam: Deep visual slam for monocular, stereo, and rgb-d cameras","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.040954Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:87cecc8f658e90fe839924a40149a7abf946d228631fc245cd27176c761f06f6","observation_id":"6e54c878-c6ec-4f94-bafe-311748d59852","resolution":{"observed_at":"2026-08-12T21:07:40.040954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.858309Z","title":"Cutting edge, flagship camera with intelli- gent feedback and resolution","venue":null,"work_id":"fbebe31e-4224-4db6-981f-c7802d1e83fd","year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.049599Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:abb1e5d5cbf28aebf0f0d13faec3114e58b0040576429c2a7981fb6230e416e1","observation_id":"72f32c76-1d00-4967-aa7a-dc851eeec0bd","resolution":{"observed_at":"2026-08-12T21:07:40.862680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.846358Z","title":"Attention is all you need","venue":null,"work_id":"74db1b99-e8d2-47ec-b792-4c101de677c5","year":2017},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.053366Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:83b956089747c56cc1a82bee72ee072510079ae498c8627fd7c07c9db5967d56","observation_id":"f643168a-a8de-4260-a3c0-17f713032efe","resolution":{"observed_at":"2026-08-12T21:07:40.850627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.833626Z","title":"Three-dimensional scene flow","venue":null,"work_id":"495643ac-fb2f-4aa2-b5be-92817815f0fe","year":1999},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.057335Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:4bf3cdb1b2f5929429183cb53431f0c4f63e5e8e400eb8055c6808db55323e0c","observation_id":"ec255f23-e6be-4892-b366-36486d49a1c3","resolution":{"observed_at":"2026-08-12T21:07:40.837299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16061","last_updated":"2024-08-28T18:01:00Z","snapshot_observed_at":"2026-08-06T09:17:41.634492Z","submitted_at":"2024-08-28T18:01:00Z","title":"3D Reconstruction with Spatial Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16061","snapshot_observed_at":"2026-08-12T21:07:40.061138Z","title":"3d reconstruction with spatial memory","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.061138Z"},"links":{"cited_paper":"/paper/2408.16061","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:c96990768ffb0a592925b736688cc78a92c8ee14796039b54d2895230253350f","observation_id":"adccae37-4488-41c4-a8ee-9ac42fc14826","resolution":{"observed_at":"2026-08-12T21:07:40.061138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.821888Z","title":"Vggsfm: Visual geometry grounded deep structure from motion","venue":null,"work_id":"3011c1d2-c10c-4e25-a683-705947c801a2","year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.065433Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:d0539313ad0d203035ff89eb012548b463e829b9a8bce78a5aa176a4a6292707","observation_id":"3d66aab6-8614-482b-aeee-a46ed14eb667","resolution":{"observed_at":"2026-08-12T21:07:40.825954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.069360Z","title":"Shape of motion: 4d reconstruction from a single video","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.069360Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:648171fedf59c6846e3b51eeb71a26a82d3186758aba73fbee4033b3a6a51337","observation_id":"94413edf-b1e1-4f10-8258-4c030f14e467","resolution":{"observed_at":"2026-08-12T21:07:40.069360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12387","last_updated":"2025-01-21T18:59:23Z","snapshot_observed_at":"2026-08-10T17:10:19.492897Z","submitted_at":"2025-01-21T18:59:23Z","title":"Continuous 3D Perception Model with Persistent State","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12387","snapshot_observed_at":"2026-08-12T21:07:40.073445Z","title":"Continuous 3d perception model with persistent state","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.073445Z"},"links":{"cited_paper":"/paper/2501.12387","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:d7dc2ddb4c31a946565ebc5fcc8eb6b56a86ab89739582d724c0ca5b8bf7c7d8","observation_id":"cf6e0524-e375-4568-af28-06c2946f83f3","resolution":{"observed_at":"2026-08-12T21:07:40.073445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.810420Z","title":"Pov-surgery: A dataset for egocentric hand and tool pose estimation during surgi- cal activities","venue":null,"work_id":"6072468b-c00d-4337-bebf-94c1eda1ab30","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.077918Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:63ffa89498d11575b36098a144380edcff7294bd2a40837b2d9096d5fa315948","observation_id":"a60015aa-3f49-48dc-88f5-bc7ec2655e6b","resolution":{"observed_at":"2026-08-12T21:07:40.814207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.798141Z","title":"Dust3r: Geometric 3d vision made easy","venue":null,"work_id":"1021c836-f3cb-4240-9246-e78287834e7e","year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.081702Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:a24553c01ef9929fdd9815f7e49d6e3da25dcb3c4dd902cfafa7c3e2234df4aa","observation_id":"b53426f0-28df-447c-aa8a-e6b6383b7121","resolution":{"observed_at":"2026-08-12T21:07:40.801928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.786713Z","title":"Tar- tanvo: A generalizable learning-based vo","venue":null,"work_id":"07b7a858-5b92-4b7f-8b63-e96c939dd634","year":2021},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.085044Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:6c3fcd13f03fa5624e584e0005d725e2e90cc77f73dd7ae89046d1f5f99ebf31","observation_id":"c399aa9f-7d85-4bea-a234-cae103c88bd2","resolution":{"observed_at":"2026-08-12T21:07:40.790517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.089720Z","title":"Neural video depth stabilizer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.089720Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:6280ea36981f60709a191012d08a1833efe932bc0a932abcf99b1561a79a82cf","observation_id":"fcd80a5a-f3f3-490a-b3dc-aa92790581ce","resolution":{"observed_at":"2026-08-12T21:07:40.089720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.767185Z","title":"Egocentric video comprehension via large language model inner speech","venue":null,"work_id":"9e61a174-ca83-485a-a543-de7e17b350df","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.094031Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:5f1e0951ddb68e9bbbe3b475ca728c1a596d2ff319691faeb527d077ce8efdb9","observation_id":"0aefd348-9a0c-4673-b6b1-1a4f21b1140a","resolution":{"observed_at":"2026-08-12T21:07:40.771100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.00025","last_updated":"2024-07-12T12:51:00Z","snapshot_observed_at":"2026-07-06T17:09:59.848387Z","submitted_at":"2023-12-28T23:34:43Z","title":"Any-point Trajectory Modeling for Policy Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.00025","snapshot_observed_at":"2026-08-12T21:07:40.097465Z","title":"Any-point trajectory modeling for policy learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.097465Z"},"links":{"cited_paper":"/paper/2401.00025","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:72ab4bfe51d7ff37ee3f3c82707e4ea96732d11c47c2c149f24ac64c2965772b","observation_id":"4f0cd970-e03d-413a-845a-b909bef233f3","resolution":{"observed_at":"2026-08-12T21:07:40.097465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12389","last_updated":"2024-11-21T20:28:33Z","snapshot_observed_at":"2026-08-13T00:27:11.273432Z","submitted_at":"2024-04-18T17:59:53Z","title":"Moving Object Segmentation: All You Need Is SAM (and Flow)","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12389","snapshot_observed_at":"2026-08-12T21:07:40.101194Z","title":"Moving object segmentation: All you need is sam (and flow)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.101194Z"},"links":{"cited_paper":"/paper/2404.12389","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:5f8132368ec9329fe9bc289983862a488969beb0c127ea26848e0235d35d2d93","observation_id":"8281ac46-3f53-4c46-906d-71ccccd1b9c5","resolution":{"observed_at":"2026-08-12T21:07:40.101194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.105977Z","title":"Gmflow: Learning optical flow via global matching","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.105977Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:86b195c22177c7125c70dddca2946fe892a64a8552d603bba5e77bcff7919a8e","observation_id":"ca0a3096-93fd-4acd-a139-2891745a6dc3","resolution":{"observed_at":"2026-08-12T21:07:40.105977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.748257Z","title":"Depth anything: Un- leashing the power of large-scale unlabeled data","venue":null,"work_id":"1af14824-3a72-44a8-8916-b0195c53db1f","year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.110069Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:853d522eca3ef9c14c6db26e0604e02a9b3d0775bca6bd7023cb128172d0b779","observation_id":"52750bf8-f045-4394-878a-3cb63653b417","resolution":{"observed_at":"2026-08-12T21:07:40.752271Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.736681Z","title":"Every pixel counts: Unsupervised geome- try learning with holistic 3d motion understanding","venue":null,"work_id":"9c45a775-bf5a-490e-a627-484305c9102c","year":2018},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.113725Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:c8f4537576b05913943b111c89842b40e2666b6308152b54163e5cd99348c46f","observation_id":"c75ec479-e73d-4e24-94ac-81ffbbd43533","resolution":{"observed_at":"2026-08-12T21:07:40.740780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.725762Z","title":"Met- ric3d: Towards zero-shot metric 3d prediction from a sin- gle image","venue":null,"work_id":"e1da2f3c-9e01-4187-a97c-3ef4b99af142","year":2023},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.117312Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:941e3175e00c1b333da05f25713aceaa7588a0aaf6b519b651b399f620f6c3e2","observation_id":"905bc388-9413-4c4c-b35f-d404585947b6","resolution":{"observed_at":"2026-08-12T21:07:40.729430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.11439","last_updated":"2024-09-23T08:22:04Z","snapshot_observed_at":"2026-08-13T04:38:23.482264Z","submitted_at":"2024-01-21T09:39:11Z","title":"General Flow as Foundation Affordance for Scalable Robot Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.11439","snapshot_observed_at":"2026-08-12T21:07:40.121084Z","title":"General flow as foundation affordance for scalable robot learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.121084Z"},"links":{"cited_paper":"/paper/2401.11439","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:8303971666f213cc3e0604f5a7c962713671dd0df5c21bd38c9e98981367f954","observation_id":"1d52056b-c92f-4d1a-95b7-543183594a55","resolution":{"observed_at":"2026-08-12T21:07:40.121084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.711577Z","title":"Recent trends in 3d reconstruction of general non-rigid scenes","venue":null,"work_id":"11526fb1-5b41-43fc-9786-677ae4460e6d","year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.125318Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:97a4e3f135da55925dc27464362c45274038a6c14297d03b90af91384df43418","observation_id":"e63e0849-55fd-436c-b399-f28d2880287c","resolution":{"observed_at":"2026-08-12T21:07:40.716617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19811","last_updated":"2024-10-02T04:00:01Z","snapshot_observed_at":"2026-08-12T23:32:55.135778Z","submitted_at":"2024-06-28T10:39:36Z","title":"EgoGaussian: Dynamic Scene Understanding from Egocentric Video with 3D Gaussian Splatting","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19811","snapshot_observed_at":"2026-08-12T21:07:40.129364Z","title":"Egogaussian: Dynamic scene understanding from egocentric video with 3d gaussian splatting","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.129364Z"},"links":{"cited_paper":"/paper/2406.19811","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:826d3c57b76826aafe6ec4813d4527801f7be477e2c415fd7e5ee30b5da293df","observation_id":"50a9096a-73c7-4311-8bb8-c8c35eba7bdf","resolution":{"observed_at":"2026-08-12T21:07:40.129364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03825","last_updated":"2025-05-08T08:32:16Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:00:07Z","title":"MonST3R: A Simple Approach for Estimating Geometry in the Presence of Motion","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03825","snapshot_observed_at":"2026-08-12T21:07:40.133398Z","title":"Monst3r: A simple approach for esti- mating geometry in the presence of motion","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.133398Z"},"links":{"cited_paper":"/paper/2410.03825","citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:4500da344544278aec758adf8c69f69f9d88aa03252ca40164ddc15ab82871b8","observation_id":"073fd12c-dfd5-4ea0-8c74-0882fa75dcb2","resolution":{"observed_at":"2026-08-12T21:07:40.133398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:07:40.699775Z","title":"Fine-grained egocentric hand-object segmentation: Dataset, model, and applications","venue":null,"work_id":"5323c8fd-39e2-4e29-85a3-f3764e0b6024","year":2022},"citing_paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","version":4},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-12T21:07:40.137416Z"},"links":{"citing_paper":"/paper/2411.09145"},"observation_digest":"sha256:123cae7773d1ca1a1a570c1d8fad59acb52d059abd46326bc04e2eb4c5362f7f","observation_id":"787ee4c0-3e36-4aaa-b5f9-553c447c0cf8","resolution":{"observed_at":"2026-08-12T21:07:40.703780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.09145","last_updated":"2025-07-12T07:00:25Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T07:33:58.476366Z","submitted_at":"2024-11-14T02:57:11Z","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":70,"verified_exact":0,"verified_fuzzy":30},"total_outbound_references":105},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 100 of 105 outbound references and 1 inbound Pith citation observation for arXiv:2411.09145."}