{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FDFAB6STLEPER3C2BHMIEXC56W","short_pith_number":"pith:FDFAB6ST","schema_version":"1.0","canonical_sha256":"28ca00fa53591e48ec5a09d8825c5df58164830ab1e0886536e8e931b070f7c0","source":{"kind":"arxiv","id":"2302.14816","version":1},"attestation_state":"computed","paper":{"title":"Monocular Depth Estimation using Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhishek Kar, David J. Fleet, Mohammad Norouzi, Saurabh Saxena","submitted_at":"2023-02-28T18:08:21Z","abstract_excerpt":"We formulate monocular depth estimation using denoising diffusion models, inspired by their recent successes in high fidelity image generation. To that end, we introduce innovations to address problems arising due to noisy, incomplete depth maps in training data, including step-unrolled denoising diffusion, an $L_1$ loss, and depth infilling during training. To cope with the limited availability of data for supervised training, we leverage pre-training on self-supervised image-to-image translation tasks. Despite the simplicity of the approach, with a generic loss and architecture, our DepthGen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.14816","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-02-28T18:08:21Z","cross_cats_sorted":[],"title_canon_sha256":"5069152cc970b7d80ef2cb43924b6eb3636924e60a5f9c0fcbdf23fdd2214ca1","abstract_canon_sha256":"9e97e5c1010957834dc268777ec3dfa7fde796fde6bcdc3fef9926431ffbdd1d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:46:50.177816Z","signature_b64":"SeFuVjd8kQGgJjWrN+J1BGRv2r+xxs6TZxjFC5ggjIE5/zV4kkqm/K/W2O83SImQX2sfN8h58Z9XnScrUWLxAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"28ca00fa53591e48ec5a09d8825c5df58164830ab1e0886536e8e931b070f7c0","last_reissued_at":"2026-07-05T05:46:50.177460Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:46:50.177460Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Monocular Depth Estimation using Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abhishek Kar, David J. Fleet, Mohammad Norouzi, Saurabh Saxena","submitted_at":"2023-02-28T18:08:21Z","abstract_excerpt":"We formulate monocular depth estimation using denoising diffusion models, inspired by their recent successes in high fidelity image generation. To that end, we introduce innovations to address problems arising due to noisy, incomplete depth maps in training data, including step-unrolled denoising diffusion, an $L_1$ loss, and depth infilling during training. To cope with the limited availability of data for supervised training, we leverage pre-training on self-supervised image-to-image translation tasks. Despite the simplicity of the approach, with a generic loss and architecture, our DepthGen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.14816","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.14816/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.14816","created_at":"2026-07-05T05:46:50.177520+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.14816v1","created_at":"2026-07-05T05:46:50.177520+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.14816","created_at":"2026-07-05T05:46:50.177520+00:00"},{"alias_kind":"pith_short_12","alias_value":"FDFAB6STLEPE","created_at":"2026-07-05T05:46:50.177520+00:00"},{"alias_kind":"pith_short_16","alias_value":"FDFAB6STLEPER3C2","created_at":"2026-07-05T05:46:50.177520+00:00"},{"alias_kind":"pith_short_8","alias_value":"FDFAB6ST","created_at":"2026-07-05T05:46:50.177520+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30370","citing_title":"MUSE: Unlocking Timestep as Native Task Steering for One-Step Dense Prediction","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01099","citing_title":"Geometry-aware 4D Video Generation for Robot Manipulation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2405.10314","citing_title":"CAT3D: Create Anything in 3D with Multi-View Diffusion Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24575","citing_title":"Diffusion Model as a Generalist Segmentation Learner","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W","json":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W.json","graph_json":"https://pith.science/api/pith-number/FDFAB6STLEPER3C2BHMIEXC56W/graph.json","events_json":"https://pith.science/api/pith-number/FDFAB6STLEPER3C2BHMIEXC56W/events.json","paper":"https://pith.science/paper/FDFAB6ST"},"agent_actions":{"view_html":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W","download_json":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W.json","view_paper":"https://pith.science/paper/FDFAB6ST","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.14816&json=true","fetch_graph":"https://pith.science/api/pith-number/FDFAB6STLEPER3C2BHMIEXC56W/graph.json","fetch_events":"https://pith.science/api/pith-number/FDFAB6STLEPER3C2BHMIEXC56W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W/action/storage_attestation","attest_author":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W/action/author_attestation","sign_citation":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W/action/citation_signature","submit_replication":"https://pith.science/pith/FDFAB6STLEPER3C2BHMIEXC56W/action/replication_record"}},"created_at":"2026-07-05T05:46:50.177520+00:00","updated_at":"2026-07-05T05:46:50.177520+00:00"}