{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CZDQ3F7NARNWBPAKRYA2DOFO4J","short_pith_number":"pith:CZDQ3F7N","schema_version":"1.0","canonical_sha256":"16470d97ed045b60bc0a8e01a1b8aee253691e91ca9c2f0cc5eabd2cb2be282c","source":{"kind":"arxiv","id":"2407.19156","version":2},"attestation_state":"computed","paper":{"title":"Robust Multimodal 3D Object Detection via Modality-Agnostic Decoding and Proximity-based Modality Ensemble","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hyunwoo J. Kim, Injae Kim, Jihwan Park, Juhan Cha, Minseok Joo, Sanghyeok Lee","submitted_at":"2024-07-27T03:21:44Z","abstract_excerpt":"Recent advancements in 3D object detection have benefited from multi-modal information from the multi-view cameras and LiDAR sensors. However, the inherent disparities between the modalities pose substantial challenges. We observe that existing multi-modal 3D object detection methods heavily rely on the LiDAR sensor, treating the camera as an auxiliary modality for augmenting semantic details. This often leads to not only underutilization of camera data but also significant performance degradation in scenarios where LiDAR data is unavailable. Additionally, existing fusion methods overlook the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.19156","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-27T03:21:44Z","cross_cats_sorted":[],"title_canon_sha256":"e3157c4d4acf27d835c7b296cc523178d8d4936f5aab9cb3c6aa5d9cc53e37ac","abstract_canon_sha256":"f78dc048afa3aa0abc33c5584da7132e694ce05265a7521b6610acff5ef63e64"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:23.112751Z","signature_b64":"bG/gRAOl6CLyVM8buuJ9a6vbdkZ3/74n8mWvC9+2mc45xFLbM3QhKJJlT5w8QJFwYlXuGiQemEjpeie4naESCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16470d97ed045b60bc0a8e01a1b8aee253691e91ca9c2f0cc5eabd2cb2be282c","last_reissued_at":"2026-07-05T08:56:23.112219Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:23.112219Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robust Multimodal 3D Object Detection via Modality-Agnostic Decoding and Proximity-based Modality Ensemble","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hyunwoo J. Kim, Injae Kim, Jihwan Park, Juhan Cha, Minseok Joo, Sanghyeok Lee","submitted_at":"2024-07-27T03:21:44Z","abstract_excerpt":"Recent advancements in 3D object detection have benefited from multi-modal information from the multi-view cameras and LiDAR sensors. However, the inherent disparities between the modalities pose substantial challenges. We observe that existing multi-modal 3D object detection methods heavily rely on the LiDAR sensor, treating the camera as an auxiliary modality for augmenting semantic details. This often leads to not only underutilization of camera data but also significant performance degradation in scenarios where LiDAR data is unavailable. Additionally, existing fusion methods overlook the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.19156","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.19156/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.19156","created_at":"2026-07-05T08:56:23.112288+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.19156v2","created_at":"2026-07-05T08:56:23.112288+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.19156","created_at":"2026-07-05T08:56:23.112288+00:00"},{"alias_kind":"pith_short_12","alias_value":"CZDQ3F7NARNW","created_at":"2026-07-05T08:56:23.112288+00:00"},{"alias_kind":"pith_short_16","alias_value":"CZDQ3F7NARNWBPAK","created_at":"2026-07-05T08:56:23.112288+00:00"},{"alias_kind":"pith_short_8","alias_value":"CZDQ3F7N","created_at":"2026-07-05T08:56:23.112288+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30983","citing_title":"Can BEV Perception Gracefully Degrade under Sensor Failures?","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2409.07825","citing_title":"Deep Multimodal Learning with Missing Modality: A Survey","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J","json":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J.json","graph_json":"https://pith.science/api/pith-number/CZDQ3F7NARNWBPAKRYA2DOFO4J/graph.json","events_json":"https://pith.science/api/pith-number/CZDQ3F7NARNWBPAKRYA2DOFO4J/events.json","paper":"https://pith.science/paper/CZDQ3F7N"},"agent_actions":{"view_html":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J","download_json":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J.json","view_paper":"https://pith.science/paper/CZDQ3F7N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.19156&json=true","fetch_graph":"https://pith.science/api/pith-number/CZDQ3F7NARNWBPAKRYA2DOFO4J/graph.json","fetch_events":"https://pith.science/api/pith-number/CZDQ3F7NARNWBPAKRYA2DOFO4J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J/action/storage_attestation","attest_author":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J/action/author_attestation","sign_citation":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J/action/citation_signature","submit_replication":"https://pith.science/pith/CZDQ3F7NARNWBPAKRYA2DOFO4J/action/replication_record"}},"created_at":"2026-07-05T08:56:23.112288+00:00","updated_at":"2026-07-05T08:56:23.112288+00:00"}