{"as_of":"2026-08-20T19:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9667d9f8eab4033132b2e383082d64549d27d43dcada54ebfbaadd26eff603d6","coverage":[{"denominator":73,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":73,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T04:47:32.086124Z","state":"measured"},{"denominator":73,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":73,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.00482/citation-record","integrity":"/paper/2505.00482/integrity","json":"/paper/2505.00482/citation-record.json","paper":"/paper/2505.00482"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.354596Z","title":"Depthformer: Multi- scale vision transformer for monocular depth estimation with global local information fusion","venue":null,"work_id":"fe07031e-02a0-4344-84bf-f2eebc93d198","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.740489Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:7aefc15b007e91a184bfdab5be3ac43f05f762f75a9286a5bc3a4f00f4a38af0","observation_id":"e20639d2-b967-495e-8357-f167766ded85","resolution":{"observed_at":"2026-08-16T04:47:33.359818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.338727Z","title":"Multimae: Multi-modal multi-task masked autoen- coders","venue":null,"work_id":"ae42f742-fa9e-4fef-8bd2-34484649b77e","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.745882Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:483b08dc94f5e3dab863160d225cc8bc4e19176234e58b9d5b51474129496297","observation_id":"638bf7f6-2956-4ca9-bfa5-26245c5c6d99","resolution":{"observed_at":"2026-08-16T04:47:33.343791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.322455Z","title":"4m-21: An any-to-any vision model for tens of tasks and modalities.Advances in Neural Infor- mation Processing Systems, 37:61872–61911, 2024","venue":null,"work_id":"de5ff678-0877-4255-86cc-f4510ab75047","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.751082Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ff6f6bb78741a988000f620088748f8daf92835404f8e4c8ddb7a5eab9afd055","observation_id":"539feb88-6b35-41e6-8ce0-4251eec86a4b","resolution":{"observed_at":"2026-08-16T04:47:33.327777Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.307084Z","title":"Multidiffusion: Fusing diffusion paths for controlled image generation","venue":null,"work_id":"0363733d-2e69-4154-b546-8a9971f79105","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.756124Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:6308036eca3a130f372de62589543405e24ea6df17e3050468ef5bff72a08c54","observation_id":"cfb5ebd8-5431-4fb5-b5a0-49ba2da2c7fd","resolution":{"observed_at":"2026-08-16T04:47:33.311923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.761070Z","title":"Se- mantickitti: A dataset for semantic scene understanding of lidar sequences","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.761070Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:2d1c10dbc5090c3983e9e642a64c704cf8812b390439e418b2bc8c9a183dde8d","observation_id":"3bac5b32-4eb9-4221-bf1c-6d950f95d84e","resolution":{"observed_at":"2026-08-16T04:47:31.761070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.765916Z","title":"Loosec- ontrol: Lifting controlnet for generalized depth conditioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.765916Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:c84cd2f91fe6ee7e1d41771a54089f604d175d4ba598c65cd1852dcfbd77bbbe","observation_id":"84c0527e-61dc-4832-96bb-ae5d9348171d","resolution":{"observed_at":"2026-08-16T04:47:31.765916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.266420Z","title":"Flux.1.https://huggingface","venue":null,"work_id":"3c1b5289-2329-4827-9cfa-38d34cde4e42","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.770700Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:aba34e5eeb5610b4826ac2061f75e646b1ea615cb99cdedd2f8c27e412ea880e","observation_id":"277bc16c-9d08-438e-b729-c3ddcb67f6da","resolution":{"observed_at":"2026-08-16T04:47:33.271520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.250201Z","title":"Diffusion forcing: Next-token prediction meets full-sequence diffu- sion.Advances in Neural Information Processing Systems, 37:24081–24125, 2025","venue":null,"work_id":"8cd8ec2d-3a0a-4896-b479-9e46ce8c9dcf","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.776031Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:d01072cc2b7a225b7aaf7a556686f7c811dc5340b1e3c2d962421faf6c9333a0","observation_id":"f6408f9e-5835-4446-ac12-295d3e3fe99c","resolution":{"observed_at":"2026-08-16T04:47:33.255156Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.232077Z","title":"Text2tex: Text-driven tex- ture synthesis via diffusion models","venue":null,"work_id":"f56b3f11-3dc5-45f2-a0b0-b5a3237f5fef","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.780461Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:489fb7c4fa4e15ead33002ee08d73796127b524b67ccc73ace456faebab63571","observation_id":"a12b82ee-2136-43d4-999c-d8b68e4a8ee2","resolution":{"observed_at":"2026-08-16T04:47:33.238261Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00426","last_updated":"2023-12-29T16:42:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-30T16:18:00Z","title":"PixArt-$\\alpha$: Fast Training of Diffusion Transformer for Photorealistic Text-to-Image Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00426","snapshot_observed_at":"2026-08-16T04:47:31.785181Z","title":"Pixart-α: Fast training of diffusion transformer for photorealistic text-to-image synthesis.arXiv preprint arXiv:2310.00426, 2023","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.785181Z"},"links":{"cited_paper":"/paper/2310.00426","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:aa38c4133e7f9d521ee4b695a721d1e50eb445e2c23ffdb8f33f3e8fc272ed43","observation_id":"e3322d9e-bd64-4fc3-91e3-778378b88a1f","resolution":{"observed_at":"2026-08-16T04:47:31.785181Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.216606Z","title":"Neural ordinary differential equa- tions","venue":null,"work_id":"404d2840-84f2-4368-9d20-dabfbfc3a1af","year":2018},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.790267Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:456c420361049c63b7140d583d4045675c9764879af78a18625122d5142464d7","observation_id":"e424f2f6-1aae-42a9-a2ce-834c84b7f418","resolution":{"observed_at":"2026-08-16T04:47:33.221679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.200500Z","title":"Deep diffusion image prior for efficient ood adaptation in 3d inverse problems","venue":null,"work_id":"ea15e3a1-e8c0-4e3d-8663-208b8f8d3d64","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.795385Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:0b8d0179eeba2990593ab211eb10faa7b9296894c85cf720f79b8e5fb5d0e5ab","observation_id":"6ab98a38-4b5a-4ca4-8e11-d3c8ca25a96f","resolution":{"observed_at":"2026-08-16T04:47:33.205553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.183060Z","title":"Improving diffusion models for inverse prob- lems using manifold constraints.Advances in Neural Infor- mation Processing Systems, 35:25683–25696, 2022","venue":null,"work_id":"23126879-83b6-4c36-9d8a-72fb5625d563","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.800111Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f9c7e6b7b70c1081d1fde1e90fd1e1950e7fd724305a33878a866c83157605ec","observation_id":"5b0afb4f-3199-4fbb-b843-be69d0908e23","resolution":{"observed_at":"2026-08-16T04:47:33.188360Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.804493Z","title":"Solving 3d inverse problems us- ing pre-trained 2d diffusion models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.804493Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:6604725791a4e7b99b4d1fb6b0812834d44ee3efe8947b40b17ccbe407332c7a","observation_id":"44d0262f-e37b-4fe7-931f-c3215bc8e681","resolution":{"observed_at":"2026-08-16T04:47:31.804493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.153730Z","title":"Latentpaint: Image inpainting in latent space with diffusion models","venue":null,"work_id":"3933dcd4-6d2c-4037-b176-9407ab3558a5","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.809715Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:3386f919dc790928d8c0c553b673abdbe8448e7ab40472d116638c349a85a98b","observation_id":"0a47916b-fd92-4230-8d66-d34143d1b414","resolution":{"observed_at":"2026-08-16T04:47:33.158873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11427","last_updated":"2022-10-20T17:16:37Z","snapshot_observed_at":"2026-08-16T16:22:28.831035Z","submitted_at":"2022-10-20T17:16:37Z","title":"DiffEdit: Diffusion-based semantic image editing with mask guidance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.11427","snapshot_observed_at":"2026-08-16T04:47:31.814438Z","title":"Diffedit: Diffusion-based seman- tic image editing with mask guidance.arXiv preprint arXiv:2210.11427, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.814438Z"},"links":{"cited_paper":"/paper/2210.11427","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:aad4facd5428b805e3a298e97d81c3d4bda570a2957c086b1640904c3bf273ea","observation_id":"d25b22bf-6406-464c-ae7c-b8a6b27e4b97","resolution":{"observed_at":"2026-08-16T04:47:31.814438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.136977Z","title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes","venue":null,"work_id":"4eaefb08-fd3d-4e0c-b48a-ec337480e0ea","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.819346Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ee55cf3e3338772c6681bcbed54cceb5c53232bc927ce509856c56be6280fb10","observation_id":"7508e38b-4b78-4cd9-a750-98697494ab7b","resolution":{"observed_at":"2026-08-16T04:47:33.142025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.118794Z","title":"Scaling vision transformers to 22 billion pa- rameters","venue":null,"work_id":"ae6e9d2b-e854-4568-8e7e-33f157eabb96","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.823696Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:4ac43605a5f81a5538b3c4ddd6f5540b872bef57403ec20ffa84510649187039","observation_id":"f77ecb42-ba28-4636-8774-acdf2170ad25","resolution":{"observed_at":"2026-08-16T04:47:33.124854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.099455Z","title":"Scaling rec- tified flow transformers for high-resolution image synthesis","venue":null,"work_id":"c53d8a08-2ae1-4f3a-ac82-b71ce78d808c","year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.828210Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b784097797c9afcdf5e9dc05796bdba50af6b93214f1a6ba6bd6518a0f8d0497","observation_id":"d6ddda39-23d1-4c91-8f80-b2fcab394fec","resolution":{"observed_at":"2026-08-16T04:47:33.106749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.079883Z","title":"The pascal visual object classes (voc) challenge.International journal of computer vision, 88:303–338, 2010","venue":null,"work_id":"c5143477-28e7-4fb9-a80d-6c60e7bc2ad6","year":2010},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.832964Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:d2ecc5ebf27705efdf757b8bae9756ef1e21d482dcfbff578116392effc3725f","observation_id":"88ba2f57-a753-471b-9475-c7cb6b4f5e86","resolution":{"observed_at":"2026-08-16T04:47:33.084975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.064667Z","title":"Geowiz- ard: Unleashing the diffusion priors for 3d geometry esti- mation from a single image","venue":null,"work_id":"ca4635aa-9c03-4dba-a452-c902ce95c001","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.837638Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:6e342a7ebd8929aa464502c89640496af61d122fa14a8f96f2c10c6ddcccdb5e","observation_id":"184ad83e-704c-4dc8-8e76-83fd531d14c9","resolution":{"observed_at":"2026-08-16T04:47:33.069501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.048629Z","title":"Fischer, Ulrich Prestel, Pingchuan Ma, Dmytro Kotovenko, Olga Grebenkova, Stefan Andreas Baumann, Vincent Tao Hu, and Bj ¨orn Ommer","venue":null,"work_id":"91c158d9-7619-4b15-a665-a567933b256c","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.842853Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:8456ed832eefee7e410f56985d07336025493bbfc7cddaa1e672cff9aaf18330","observation_id":"34108945-3372-47fa-9433-be9dffdcba9b","resolution":{"observed_at":"2026-08-16T04:47:33.053680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.033286Z","title":"Efficient diffu- sion training via min-snr weighting strategy","venue":null,"work_id":"7b72e781-aac0-4928-9b56-4f827052d2e1","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.847919Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:67aa23553b7b7bab31b427428c0c33d66bc2256a5e571f8bef4b8906ee905ffb","observation_id":"70114f14-a7c6-4740-9e1e-5bf642aac3bb","resolution":{"observed_at":"2026-08-16T04:47:33.038231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:33.014331Z","title":"Gans trained by a two time-scale update rule converge to a local nash equilib- rium.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":"3fdfefb8-4802-44e9-8b71-dc2a93442801","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.852592Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b1e37de18d80a434999023c5144ce8d73ec19a79bc890942098e172887602aa7","observation_id":"bab29a7a-832b-4f96-9616-516c93de7b41","resolution":{"observed_at":"2026-08-16T04:47:33.022264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.12598","last_updated":"2022-07-26T01:42:07Z","snapshot_observed_at":"2026-08-14T06:37:15.299690Z","submitted_at":"2022-07-26T01:42:07Z","title":"Classifier-Free Diffusion Guidance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.12598","snapshot_observed_at":"2026-08-16T04:47:31.857329Z","title":"Classifier-free diffusion guidance.arXiv preprint arXiv:2207.12598, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.857329Z"},"links":{"cited_paper":"/paper/2207.12598","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:e0ada06fccf70876c0a8ce36dcf245356796d1b63bea3af689f276a4656fd4a7","observation_id":"b6b1a09d-5fe0-40f2-817d-6c5c76184e18","resolution":{"observed_at":"2026-08-16T04:47:31.857329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.862188Z","title":"Denoising dif- fusion probabilistic models.Advances in neural information processing systems, 33:6840–6851, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.862188Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:26473bd43377625492657c42825891cfaff4b610536d6217bf444b99b3e9592b","observation_id":"7a2cdd4b-fcd0-4adf-b77c-fdc4440b7dd4","resolution":{"observed_at":"2026-08-16T04:47:31.862188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.984592Z","title":"Lora: Low-rank adaptation of large language models.ICLR, 1(2):3, 2022","venue":null,"work_id":"5442bc74-c793-40b2-89ed-80e9a202838e","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.866706Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:9bc0edd30a7b89c76d08ef18a82232d23b7c972ee8506807f1163683371fa357","observation_id":"f12d4fa1-d866-45ac-9bdd-a24740c0f68b","resolution":{"observed_at":"2026-08-16T04:47:32.990748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.967091Z","title":"Zero-shot depth completion via test-time align- ment with affine-invariant depth prior","venue":null,"work_id":"e890e267-58a6-4182-9797-a1312b7fb4fe","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.871529Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:06daa38a1a250f694d818995ae688834832502923cc342e3513ded1247c82f67","observation_id":"2b127d2a-7d10-404e-bd23-f06a61b4a54e","resolution":{"observed_at":"2026-08-16T04:47:32.972436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.02412","last_updated":"2023-02-05T15:49:26Z","snapshot_observed_at":"2026-08-19T16:26:36.246714Z","submitted_at":"2023-02-05T15:49:26Z","title":"Mixture of Diffusers for scene composition and high resolution image generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.02412","snapshot_observed_at":"2026-08-16T04:47:31.875973Z","title":"Mixture of diffusers for scene composition and high resolution image generation.arXiv preprint arXiv:2302.02412, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.875973Z"},"links":{"cited_paper":"/paper/2302.02412","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:1d620d4d1545e077bb7230c23a3f79c9b7fe6e1ab8bb2269eb1bc2b2cc3b9020","observation_id":"8d32c6f0-a73f-4804-b121-2370953719a9","resolution":{"observed_at":"2026-08-16T04:47:31.875973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.950956Z","title":"Dy- namicstereo: Consistent dynamic depth from stereo videos","venue":null,"work_id":"743f2bf3-48e8-49df-b7e2-475f806457fd","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.881223Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ad56d59d0427f7b277dd0ada851f2ced799d6b22bd7df5689ac3a80aec574d42","observation_id":"46006c99-c6b8-431d-b66d-c74fba81ad8c","resolution":{"observed_at":"2026-08-16T04:47:32.956075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.885577Z","title":"Imagic: Text-based real image editing with diffusion models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.885577Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ac55223825af754fa5c55a15b4bcf5558d013e03742c03a69d2fc03410cca002","observation_id":"a0db4282-6c47-4564-b9b0-aba4f28d0466","resolution":{"observed_at":"2026-08-16T04:47:31.885577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.924102Z","title":"Repurpos- ing diffusion-based image generators for monocular depth estimation","venue":null,"work_id":"147d4451-969a-40d4-a9d7-f4c1c19b46cf","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.890786Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:3ba64cb4152a2a41f9e20198cf7d7f87160be06a28d31a1b120adca1e43c3ccf","observation_id":"48ec99ae-962e-4da9-843e-006bf6899dc1","resolution":{"observed_at":"2026-08-16T04:47:32.929179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.908145Z","title":"Openimages: A public dataset for large-scale multi-label and multi-class im- age classification.Dataset available from https://github","venue":null,"work_id":"a7f28973-075d-4f28-b782-d016e321b019","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.895569Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ae54fac5ae803891237f5ad16d001d497112211eaaed84b7ca8000e49b444adf","observation_id":"f05808a1-c3c4-4c9c-b47a-ef711f6c1036","resolution":{"observed_at":"2026-08-16T04:47:32.913766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.900304Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.900304Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:071c608ca6924bc59cbae3c1131b2da7f3f9199060755de3c40088a179d370b9","observation_id":"83f18812-e8e0-41ca-9a8e-bcc549e6a6ee","resolution":{"observed_at":"2026-08-16T04:47:31.900304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.879883Z","title":"A sim- ple approach to unifying diffusion-based conditional gener- ation","venue":null,"work_id":"78277890-0970-4ecd-a281-f73aac306918","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.904970Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:55be925bd96f84fd1d9f3132f637e83b29d1d26aacab8f0380fd20c527b7d68b","observation_id":"cd96c19c-110a-46e9-82c2-97154f571b78","resolution":{"observed_at":"2026-08-16T04:47:32.885523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.860275Z","title":"Matrixcity: A large-scale city dataset for city-scale neural rendering and beyond","venue":null,"work_id":"217207fd-39cd-4ca2-9b6d-bf4c364fb65b","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.910360Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:94e13f5338e0ddc84d5434d3b64d033cebc528dea871369f09d707e65b8e0c51","observation_id":"e76b6efc-0845-4b3f-bbe2-6e796cae985d","resolution":{"observed_at":"2026-08-16T04:47:32.867109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.841267Z","title":"Revisiting stereo depth estimation from a sequence- to-sequence perspective with transformers","venue":null,"work_id":"23a14d96-e68e-49a0-a740-c1b76ce0010d","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.915037Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:3066c0751afe76d8865f0d28543d8510b07170ace6f2eb5cd314712ee08dce8f","observation_id":"fc484d44-5e99-4cf7-b3ae-b2a343c091cb","resolution":{"observed_at":"2026-08-16T04:47:32.847315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.824944Z","title":"Microsoft coco: Common objects in context","venue":null,"work_id":"98b33322-1092-41b9-a4d9-ef748244957a","year":2014},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.919396Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:815ee359ae40d6c4e332da9900a6adbc4734e0c932c1a0c9a67246266f5b4ce9","observation_id":"08077ad5-380a-4e0f-9a05-2a2f72ba0fdd","resolution":{"observed_at":"2026-08-16T04:47:32.829741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02747","last_updated":"2023-02-08T15:46:05Z","snapshot_observed_at":"2026-08-16T02:30:42.660030Z","submitted_at":"2022-10-06T08:32:20Z","title":"Flow Matching for Generative Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.02747","snapshot_observed_at":"2026-08-16T04:47:31.923802Z","title":"Flow matching for generative mod- eling.arXiv preprint arXiv:2210.02747, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.923802Z"},"links":{"cited_paper":"/paper/2210.02747","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:0f8d43ab4fd85cddf69eec01d075f22ec09cf3822d2d962f13bf8b03db742d2c","observation_id":"a80c5ed8-f222-492f-b0c2-8904a0560559","resolution":{"observed_at":"2026-08-16T04:47:31.923802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.807802Z","title":"Visual instruction tuning.Advances in neural information processing systems, 36:34892–34916, 2023","venue":null,"work_id":"1cd961ee-4dbd-48f4-b7fa-504690172d2e","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.929218Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:6e1813800aaa6262073bb5b31a76ad649cc8ef03df2ada1d15e036900383881d","observation_id":"7470cf35-68c6-4d78-b1df-11819e2f20c7","resolution":{"observed_at":"2026-08-16T04:47:32.813347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.789396Z","title":"Zero-1-to- 3: Zero-shot one image to 3d object","venue":null,"work_id":"50270eb3-2ada-48c3-8901-6b013ee1edaf","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.934046Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:dd7c1133fb184509e288ad2a84604e78388de63f6bd33348943a9daa6c8cb622","observation_id":"698df1ed-81f7-4a64-9031-172d40c1a499","resolution":{"observed_at":"2026-08-16T04:47:32.794549Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.08916","last_updated":"2022-10-04T22:37:32Z","snapshot_observed_at":"2026-08-20T10:02:52.933360Z","submitted_at":"2022-06-17T17:53:47Z","title":"Unified-IO: A Unified Model for Vision, Language, and Multi-Modal Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.08916","snapshot_observed_at":"2026-08-16T04:47:31.938971Z","title":"Unified-io: A unified model for vision, language, and multi-modal tasks.arXiv preprint arXiv:2206.08916, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.938971Z"},"links":{"cited_paper":"/paper/2206.08916","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:2e1024828ff329e694e887b213e524063b1020c522f82b4f14c9ff85cdb06470","observation_id":"6fbf4387-b655-4f54-884a-02153e21f2d8","resolution":{"observed_at":"2026-08-16T04:47:31.938971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.773251Z","title":"Unified-io 2: Scaling autoregressive multimodal models with vision language audio and action","venue":null,"work_id":"1b7bd34e-dde0-46f7-8cd7-8a2fb7e53262","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.944083Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:62a2296061957cbd7eb508cb81f819106050e49a318ea3e0d60c6784e89f1e18","observation_id":"3ccc817f-b3bc-4391-bfa2-8f67f3c91286","resolution":{"observed_at":"2026-08-16T04:47:32.778339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.948760Z","title":"Repaint: Inpainting using denoising diffusion probabilistic models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.948760Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b498b734a6ae873ca081cae285380a4643628bd4aa740110501695d6a08d07ed","observation_id":"b6f02291-c849-4ad9-9595-ddabf8f720c2","resolution":{"observed_at":"2026-08-16T04:47:31.948760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.744195Z","title":"Readout guidance: Learning con- trol from diffusion features","venue":null,"work_id":"27a73783-afd2-4f07-8718-b0e27e48a7f0","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.953585Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:96af08559814981e802478e3f3dac0684fd299e1ef91c7537be7e2a6f6ab6630","observation_id":"9b57c80d-87e4-499f-b8a8-4fd6d514eea3","resolution":{"observed_at":"2026-08-16T04:47:32.750508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:31.958883Z","title":"Scalable diffusion models with transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.958883Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:41ecc6babb0c5dd280eea825d87dfd9b38658d11db88a003bacf338a4a297c66","observation_id":"9180c32e-ce26-43b9-95d9-8ec625af36ce","resolution":{"observed_at":"2026-08-16T04:47:31.958883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.711246Z","title":"Pexels, royalty-free stock footage website.https: //www.pexels.com","venue":null,"work_id":"fc7f2f7f-dae6-4cde-814d-c8a09bf54e6b","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.963503Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:68d411828ed144d930da37c190f409ec0c367025519faaf396de989eb5952cdd","observation_id":"0e50cd36-d621-4350-8f5d-8dc2b3f090b5","resolution":{"observed_at":"2026-08-16T04:47:32.717285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.695205Z","title":"Learning transferable visual models from natural language supervi- sion","venue":null,"work_id":"d836239e-5747-428d-9b8b-4469d8b2fc98","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.968322Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:bc1c64983ff90b316375a6b027e412f8a8c641c3b3a9cd8bbd3100098b2e3b75","observation_id":"6bdc975b-33d0-4546-ade7-154e0eed6cea","resolution":{"observed_at":"2026-08-16T04:47:32.700647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.680436Z","title":null,"venue":null,"work_id":"53c461d9-b235-4091-b690-eb3168cf9679","year":2020},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.972585Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:18d3606d551ca49bb6b4df96e8bfeff5174fbb6c8dfc9439ca6afade2e7d19af","observation_id":"08656bfd-de6b-4f5b-aabe-23ef6ff4240b","resolution":{"observed_at":"2026-08-16T04:47:32.685202Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.663146Z","title":"Vi- sion transformers for dense prediction","venue":null,"work_id":"80e987eb-4980-45de-93a6-c1449b3b1c6c","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.977327Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:de73e171cfdf39d21ab4e1208fd239d139747270e0c58d58757c05b5412ce29a","observation_id":"6807bea5-e543-4fd7-9e27-1ac9e589dc5f","resolution":{"observed_at":"2026-08-16T04:47:32.668627Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.647323Z","title":"Hypersim: A photorealistic syn- thetic dataset for holistic indoor scene understanding","venue":null,"work_id":"d2c166ad-07ed-4adc-9ac8-6c292a94327e","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.981665Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b76626870022d88a5630311da06a58b400a6cc3be12c539cf4970acf7c08111a","observation_id":"27fa0ec4-24af-4d52-b849-db6ee5980a96","resolution":{"observed_at":"2026-08-16T04:47:32.652645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.630227Z","title":"High-resolution image synthesis with latent diffusion models","venue":null,"work_id":"3453fde6-c6d5-4754-81f3-5e87f52ccca2","year":2022},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.986378Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:fab945c6f36e35ed98d92ab239bcf7595980209bfd26d8e871f8e8cf1ce0dff6","observation_id":"c03e09bd-8d5f-44ac-ada3-fc50d73bfa3b","resolution":{"observed_at":"2026-08-16T04:47:32.635760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.613106Z","title":"Imagenet large scale visual recognition challenge.International journal of computer vision, 115:211–252, 2015","venue":null,"work_id":"ae0a1f74-95f3-4d13-b01d-094bbdffb9ea","year":2015},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.990852Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:d3914f66aa1e8555353b02b723293ce9225ce7dfc842c634f18d2cb3bab6dd14","observation_id":"f852eb49-cb3a-4f9b-876d-7de63aafd134","resolution":{"observed_at":"2026-08-16T04:47:32.619105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.595068Z","title":"Improved techniques for training gans.Advances in neural information processing systems, 29, 2016","venue":null,"work_id":"85473153-33c6-4db5-80d0-0480b0541ff7","year":2016},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:31.995542Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b614f0d33079899a33fd1561c63c48d2f7a27bf95e9568deb6449b7fe957d6e8","observation_id":"fe6d243f-9a3d-4f55-abd5-1c4e86eedf5d","resolution":{"observed_at":"2026-08-16T04:47:32.600796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.576556Z","title":"A multi-view stereo benchmark with high- resolution images and multi-camera videos","venue":null,"work_id":"06ca7769-d1a0-4b6e-bcda-47d59f9fea99","year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.000071Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:a2ce9582fd995b7ebdb3c18d8bf53287f12652eacbdeedc95f3be9948f564367","observation_id":"0d498dfd-7bad-4841-8be7-3e54115a63db","resolution":{"observed_at":"2026-08-16T04:47:32.582793Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.557645Z","title":"Indoor segmentation and support inference from rgbd images","venue":null,"work_id":"896929d7-3238-47d4-a2d1-2932f163773d","year":2012},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.004897Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:2fcc07fde60e0ec5ec021580bf9991d2f7890d8815674bbc3cd8340f02a313b0","observation_id":"43486afc-c5c8-41fe-8f25-6cb4cb57ae12","resolution":{"observed_at":"2026-08-16T04:47:32.563874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.540938Z","title":"Generative modeling by esti- mating gradients of the data distribution.Advances in neural information processing systems, 32, 2019","venue":null,"work_id":"c339d6d8-9a7d-4af0-8d13-6bde696451b7","year":2019},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.010202Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:b7bd09dfd0a9fc7f23d98ab6950f9f7ce9059de2a7f1f5e747b40c22f0c1d4c5","observation_id":"dc3e3101-6f71-4093-851b-55de20ba80a1","resolution":{"observed_at":"2026-08-16T04:47:32.546773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.523338Z","title":"Improved techniques for training score-based generative models.Advances in neural information processing systems, 33:12438–12448, 2020","venue":null,"work_id":"8de54e05-492c-4deb-a9a3-e8d94b822e79","year":2020},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.014636Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:17bed4b50f6740e02130edc3ac3b5f6a1fb768fe6c634be12cdce5168e4ea607","observation_id":"1c97dfc6-f6ac-4964-904f-ac71d1c4f08d","resolution":{"observed_at":"2026-08-16T04:47:32.528974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.019532Z","title":"Score-based generative modeling through stochastic differential equa- tions","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.019532Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:0aed1e038d72973af44facab07de715b7b1c9720ab5106af129a970406503745","observation_id":"164c3f2f-ccec-4470-842d-d00f27b0c700","resolution":{"observed_at":"2026-08-16T04:47:32.019532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10853","last_updated":"2023-05-21T20:26:30Z","snapshot_observed_at":"2026-08-20T15:04:43.327967Z","submitted_at":"2023-05-18T10:15:06Z","title":"LDM3D: Latent Diffusion Model for 3D","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.10853","snapshot_observed_at":"2026-08-16T04:47:32.024582Z","title":"Ldm3d: Latent diffusion model for 3d.arXiv preprint arXiv:2305.10853,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.024582Z"},"links":{"cited_paper":"/paper/2305.10853","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:93b14ca843472b7312f95373630c17fb2878388ba6e96eeb3ffff62370553dd1","observation_id":"02f508d3-5f8b-45e4-b086-ada5b5d12922","resolution":{"observed_at":"2026-08-16T04:47:32.024582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06209","last_updated":"2024-12-09T05:04:50Z","snapshot_observed_at":"2026-08-20T00:57:58.550987Z","submitted_at":"2024-12-09T05:04:50Z","title":"Sound2Vision: Generating Diverse Visuals from Audio through Cross-Modal Latent Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06209","snapshot_observed_at":"2026-08-16T04:47:32.030323Z","title":"Sound2vision: Generating diverse visuals from au- dio through cross-modal latent alignment.arXiv preprint arXiv:2412.06209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.030323Z"},"links":{"cited_paper":"/paper/2412.06209","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:fb1efa9d85a59d2420423638460fd3a43cc136cadcf4e06152a9611fc190f0da","observation_id":"52086cdb-08af-45e3-8b09-c4ecca06e572","resolution":{"observed_at":"2026-08-16T04:47:32.030323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.497896Z","title":"Soundbrush: Sound as a brush for visual scene editing","venue":null,"work_id":"573bd3e0-9101-4698-8ede-8015c17d1894","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.035308Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:3aae9355eb67c298c4d2ecf64168d5975a16ad6119fd112dad90c4907076d755","observation_id":"8b9390b1-9fc4-41f2-a5e7-c474b946b25b","resolution":{"observed_at":"2026-08-16T04:47:32.503412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.480652Z","title":"Plug-and-play diffusion features for text-driven image-to-image translation","venue":null,"work_id":"48d8438e-8f78-4593-816d-2a9973bf1491","year":1921},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.040085Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f24ad81696825e46c3e81f01097bee237dc6c5f494516ddaa77fac4446e12fb9","observation_id":"e47614c9-1030-4b72-98f6-d4312ed2ba1c","resolution":{"observed_at":"2026-08-16T04:47:32.486346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.00463","last_updated":"2019-08-29T04:17:49Z","snapshot_observed_at":"2026-08-15T10:05:41.585216Z","submitted_at":"2019-08-01T15:39:54Z","title":"DIODE: A Dense Indoor and Outdoor DEpth Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.00463","snapshot_observed_at":"2026-08-16T04:47:32.044798Z","title":"Diode: A dense indoor and outdoor depth dataset.arXiv preprint arXiv:1908.00463, 2019","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.044798Z"},"links":{"cited_paper":"/paper/1908.00463","citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:753d1a5236aad367d8326087af5e1f1d92db7bba8a51f09f266b80bd22a833ac","observation_id":"7064dcd2-c2b2-4a6e-a4f7-ed06b613ff2a","resolution":{"observed_at":"2026-08-16T04:47:32.044798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.050298Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.050298Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:96c5ed731c4d81fce8a0f9fb919d5a8eb69a2b270dc70b777177634b5bc8ab86","observation_id":"d40fd372-e52c-4e32-b2af-708f6e2f0ac4","resolution":{"observed_at":"2026-08-16T04:47:32.050298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.454289Z","title":"Irs: A large naturalistic indoor robotics stereo dataset to train deep models for dis- parity and surface normal estimation","venue":null,"work_id":"b85bbe8e-cc57-430c-8b87-1fa69169d4dd","year":2021},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.055018Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:488c4a35a0711503023a847adb78b53e5066e845a6a8c4cca273f1500fd5aeb3","observation_id":"d941c7ab-f341-46d4-a091-db4be8757017","resolution":{"observed_at":"2026-08-16T04:47:32.460079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.436989Z","title":"Imagere- ward: Learning and evaluating human preferences for text- to-image generation.Advances in Neural Information Pro- cessing Systems, 36:15903–15935, 2023","venue":null,"work_id":"f93a8106-e2c3-47c9-b6f9-a11b88434cba","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.059412Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:f3e3e19a34e9ca8908a4e850be771b31419efe32d71da5a206300e402ea4c4d7","observation_id":"70dec2f1-5e85-4e52-a191-5b990e93bf10","resolution":{"observed_at":"2026-08-16T04:47:32.443142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.418248Z","title":"Depth any- thing v2.Advances in Neural Information Processing Sys- tems, 37:21875–21911, 2025","venue":null,"work_id":"c0a91945-931a-4a0b-8bd3-a9b329d1437b","year":2025},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.064200Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:7d940f44e6ac68a1821cea8ab4208e42521bd077f1029a54f811fbf0ad1f1a44","observation_id":"b913b190-caf6-4d10-ac67-96ebffd7c4a4","resolution":{"observed_at":"2026-08-16T04:47:32.424683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.399047Z","title":"Paint- it: Text-to-texture synthesis via deep convolutional texture map optimization and physically-based rendering","venue":null,"work_id":"a142aa4c-acea-4e37-aff4-da02d5bc4e25","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.068423Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:d5d10698f0894786267481eadddf194754a123cbcb9a41f699a98c0e97ea6190","observation_id":"78552e6d-0d9a-4028-bcda-1cd8f69fa44e","resolution":{"observed_at":"2026-08-16T04:47:32.404820Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.381275Z","title":"Metta: Single-view to 3d textured mesh reconstruction with test-time adaptation","venue":null,"work_id":"7b7ba2a1-3c2a-435c-a48f-7c91b0249328","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.072809Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ad2217dfb89654dd48e2c7a7d8fbb44a802b355981fc17cb5dfb7bba6040ecd7","observation_id":"061a8559-6029-4eb0-b3fd-39aa4cc40872","resolution":{"observed_at":"2026-08-16T04:47:32.386796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.364064Z","title":"Joint- net: Extending text-to-image diffusion for dense distribution modeling","venue":null,"work_id":"2e6b1395-add5-4720-a167-febcfcd7882a","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.077128Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:ed2cd2df186906615d7207974b16fba2a1050bac6f24f8f80973f2e283499a54","observation_id":"66dc40ec-204f-4850-be39-b048a42aa84f","resolution":{"observed_at":"2026-08-16T04:47:32.370386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.345001Z","title":"Adding conditional control to text-to-image diffusion models","venue":null,"work_id":"8a73d1a5-8d9a-4643-9666-f1469993c326","year":2023},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.081667Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:e6c66c254b0985d8dbbd525acedbf00c536a7d2012e5ee1ae29bb11ba44d326c","observation_id":"0ad6a0fd-d2b8-47e9-98b0-17b328268a18","resolution":{"observed_at":"2026-08-16T04:47:32.351727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"5330.7926","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:47:32.182605Z","title":"TNBMM CMBDL LJUUFO CBMBODJOH B MFWJUBUJOH QPUJPO CPUUMF GJMMFE XJUI TIJNNFSJOH CMVF MJRVJEu t1BTUB XJUI NVTISPPNT BOE CBDPOu t","venue":null,"work_id":"bc73ec4d-13eb-4ac5-b203-67a161d42db3","year":2024},"citing_paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-16T04:47:32.086124Z"},"links":{"citing_paper":"/paper/2505.00482"},"observation_digest":"sha256:712e0121acc6b0fc84f5c25331c9d29aba8f7ef7688526d70f58fdc9d22ed7bf","observation_id":"4264956c-fc13-4bcd-a31c-4e2bc28cb58d","resolution":{"observed_at":"2026-08-16T04:47:32.191857Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.00482","last_updated":"2025-08-05T08:00:04Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-20T14:59:15.430377Z","submitted_at":"2025-05-01T12:21:23Z","title":"JointDiT: Enhancing RGB-Depth Joint Modeling with Diffusion Transformers"},"reference_resolution":{"displayed":73,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":52},"total_outbound_references":73},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 73 of 73 outbound references and 0 inbound Pith citation observations for arXiv:2505.00482."}