{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RAZ75KHTHJGWEAZRP3EXHW6W7V","short_pith_number":"pith:RAZ75KHT","schema_version":"1.0","canonical_sha256":"8833fea8f33a4d6203317ec973dbd6fd733883acd7a047bbde8b5262b6bf65f6","source":{"kind":"arxiv","id":"2401.15319","version":1},"attestation_state":"computed","paper":{"title":"You Only Look Bottom-Up for Monocular 3D Object Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dingkang Liang, Dingyuan Zhang, Hongcheng Yang, Jianwei Cheng, Kaixin Xiong, Wondimu Dikubab, Xiang Bai, Zhe Liu","submitted_at":"2024-01-27T06:45:35Z","abstract_excerpt":"Monocular 3D Object Detection is an essential task for autonomous driving. Meanwhile, accurate 3D object detection from pure images is very challenging due to the loss of depth information. Most existing image-based methods infer objects' location in 3D space based on their 2D sizes on the image plane, which usually ignores the intrinsic position clues from images, leading to unsatisfactory performances. Motivated by the fact that humans could leverage the bottom-up positional clues to locate objects in 3D space from a single image, in this paper, we explore the position modeling from the imag"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.15319","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-27T06:45:35Z","cross_cats_sorted":[],"title_canon_sha256":"c559f7591ad34c536514f54c3e79dd9f2b172a6aba5dede1d2c653e49444551d","abstract_canon_sha256":"b1cd7d81c28c02848d0f491f5844c08c3d2af9bfe760cb8cd945e91ffd6c7240"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:38:14.185595Z","signature_b64":"sIj5pB5Ll3f1bCB/CSWkN14VuHW/dDGy2Jz42Laa5UKLjTD/Q4z2+ERMysXLgsxAez2IpqWATBQ0Rtof83AhBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8833fea8f33a4d6203317ec973dbd6fd733883acd7a047bbde8b5262b6bf65f6","last_reissued_at":"2026-07-05T07:38:14.185112Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:38:14.185112Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"You Only Look Bottom-Up for Monocular 3D Object Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dingkang Liang, Dingyuan Zhang, Hongcheng Yang, Jianwei Cheng, Kaixin Xiong, Wondimu Dikubab, Xiang Bai, Zhe Liu","submitted_at":"2024-01-27T06:45:35Z","abstract_excerpt":"Monocular 3D Object Detection is an essential task for autonomous driving. Meanwhile, accurate 3D object detection from pure images is very challenging due to the loss of depth information. Most existing image-based methods infer objects' location in 3D space based on their 2D sizes on the image plane, which usually ignores the intrinsic position clues from images, leading to unsatisfactory performances. Motivated by the fact that humans could leverage the bottom-up positional clues to locate objects in 3D space from a single image, in this paper, we explore the position modeling from the imag"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.15319","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.15319/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.15319","created_at":"2026-07-05T07:38:14.185174+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.15319v1","created_at":"2026-07-05T07:38:14.185174+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.15319","created_at":"2026-07-05T07:38:14.185174+00:00"},{"alias_kind":"pith_short_12","alias_value":"RAZ75KHTHJGW","created_at":"2026-07-05T07:38:14.185174+00:00"},{"alias_kind":"pith_short_16","alias_value":"RAZ75KHTHJGWEAZR","created_at":"2026-07-05T07:38:14.185174+00:00"},{"alias_kind":"pith_short_8","alias_value":"RAZ75KHT","created_at":"2026-07-05T07:38:14.185174+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.06687","citing_title":"StixelNExT++: Lightweight Monocular Scene Segmentation and Representation for Collective Perception","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V","json":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V.json","graph_json":"https://pith.science/api/pith-number/RAZ75KHTHJGWEAZRP3EXHW6W7V/graph.json","events_json":"https://pith.science/api/pith-number/RAZ75KHTHJGWEAZRP3EXHW6W7V/events.json","paper":"https://pith.science/paper/RAZ75KHT"},"agent_actions":{"view_html":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V","download_json":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V.json","view_paper":"https://pith.science/paper/RAZ75KHT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.15319&json=true","fetch_graph":"https://pith.science/api/pith-number/RAZ75KHTHJGWEAZRP3EXHW6W7V/graph.json","fetch_events":"https://pith.science/api/pith-number/RAZ75KHTHJGWEAZRP3EXHW6W7V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V/action/storage_attestation","attest_author":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V/action/author_attestation","sign_citation":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V/action/citation_signature","submit_replication":"https://pith.science/pith/RAZ75KHTHJGWEAZRP3EXHW6W7V/action/replication_record"}},"created_at":"2026-07-05T07:38:14.185174+00:00","updated_at":"2026-07-05T07:38:14.185174+00:00"}