{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:NZJTFBMLVSPLBVUMFR4HQUMVNJ","short_pith_number":"pith:NZJTFBML","schema_version":"1.0","canonical_sha256":"6e5332858bac9eb0d68c2c787851956a45dc9a794d73ec46314d92073c5b58f6","source":{"kind":"arxiv","id":"2002.12319","version":1},"attestation_state":"computed","paper":{"title":"Semantically-Guided Representation Learning for Self-Supervised Monocular Depth","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Adrien Gaidon, Jie Li, Rares Ambrus, Rui Hou, Vitor Guizilini","submitted_at":"2020-02-27T18:40:10Z","abstract_excerpt":"Self-supervised learning is showing great promise for monocular depth estimation, using geometry as the only source of supervision. Depth networks are indeed capable of learning representations that relate visual appearance to 3D properties by implicitly leveraging category-level patterns. In this work we investigate how to leverage more directly this semantic structure to guide geometric representation learning, while remaining in the self-supervised regime. Instead of using semantic labels and proxy losses in a multi-task approach, we propose a new architecture leveraging fixed pretrained se"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.12319","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-02-27T18:40:10Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"17407fb5ffec1995f8f883cb899cf769256585a8e502745a1c6475bc0e6c21b1","abstract_canon_sha256":"7da92c67a09120b174ab888f56fa8b306b9941705b593e51801f69f232832dfb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:44:22.936114Z","signature_b64":"KlFQ6yS2UR9LDqFf4Z33TsQ5s09zyZru4xss/jjAyS1N7xM0ZAU9m2S6gZaURYxWgahPCRP8pG9TrPN9fC7sBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e5332858bac9eb0d68c2c787851956a45dc9a794d73ec46314d92073c5b58f6","last_reissued_at":"2026-07-05T00:44:22.935682Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:44:22.935682Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Semantically-Guided Representation Learning for Self-Supervised Monocular Depth","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Adrien Gaidon, Jie Li, Rares Ambrus, Rui Hou, Vitor Guizilini","submitted_at":"2020-02-27T18:40:10Z","abstract_excerpt":"Self-supervised learning is showing great promise for monocular depth estimation, using geometry as the only source of supervision. Depth networks are indeed capable of learning representations that relate visual appearance to 3D properties by implicitly leveraging category-level patterns. In this work we investigate how to leverage more directly this semantic structure to guide geometric representation learning, while remaining in the self-supervised regime. Instead of using semantic labels and proxy losses in a multi-task approach, we propose a new architecture leveraging fixed pretrained se"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.12319","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.12319/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.12319","created_at":"2026-07-05T00:44:22.935743+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.12319v1","created_at":"2026-07-05T00:44:22.935743+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.12319","created_at":"2026-07-05T00:44:22.935743+00:00"},{"alias_kind":"pith_short_12","alias_value":"NZJTFBMLVSPL","created_at":"2026-07-05T00:44:22.935743+00:00"},{"alias_kind":"pith_short_16","alias_value":"NZJTFBMLVSPLBVUM","created_at":"2026-07-05T00:44:22.935743+00:00"},{"alias_kind":"pith_short_8","alias_value":"NZJTFBML","created_at":"2026-07-05T00:44:22.935743+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22686","citing_title":"SS3D: End2End Self-Supervised 3D from Web Videos","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08320","citing_title":"Improved monocular depth prediction using distance transform over pre-semantic contours with self-supervised neural networks","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22686","citing_title":"SS3D: End2End Self-Supervised 3D from Web Videos","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22686","citing_title":"SS3D: End2End Self-Supervised 3D from Web Videos","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07665","citing_title":"Adaptive Depth-converted-Scale Convolution for Self-supervised Monocular Depth Estimation","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04407","citing_title":"NAIMA: Semantics Aware RGB Guided Depth Super-Resolution","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ","json":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ.json","graph_json":"https://pith.science/api/pith-number/NZJTFBMLVSPLBVUMFR4HQUMVNJ/graph.json","events_json":"https://pith.science/api/pith-number/NZJTFBMLVSPLBVUMFR4HQUMVNJ/events.json","paper":"https://pith.science/paper/NZJTFBML"},"agent_actions":{"view_html":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ","download_json":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ.json","view_paper":"https://pith.science/paper/NZJTFBML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.12319&json=true","fetch_graph":"https://pith.science/api/pith-number/NZJTFBMLVSPLBVUMFR4HQUMVNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/NZJTFBMLVSPLBVUMFR4HQUMVNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ/action/storage_attestation","attest_author":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ/action/author_attestation","sign_citation":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ/action/citation_signature","submit_replication":"https://pith.science/pith/NZJTFBMLVSPLBVUMFR4HQUMVNJ/action/replication_record"}},"created_at":"2026-07-05T00:44:22.935743+00:00","updated_at":"2026-07-05T00:44:22.935743+00:00"}