{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ADYGT55MVFUO4QCF72UC6GPXYI","short_pith_number":"pith:ADYGT55M","schema_version":"1.0","canonical_sha256":"00f069f7aca968ee4045fea82f19f7c20bfd89c5e1ef3ca48e247b1c76b69175","source":{"kind":"arxiv","id":"2402.18140","version":1},"attestation_state":"computed","paper":{"title":"OccTransformer: Improving BEVFormer for 3D camera-only occupancy prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Borun Xu, Chuixin Kong, Donglai Wei, Jian Liu, Ruibo Ming, Sipeng Zhang, Wenyuan Zhang, Xianming Liu, Yikang Ding, Yuhang Wu","submitted_at":"2024-02-28T08:03:34Z","abstract_excerpt":"This technical report presents our solution, \"occTransformer\" for the 3D occupancy prediction track in the autonomous driving challenge at CVPR 2023. Our method builds upon the strong baseline BEVFormer and improves its performance through several simple yet effective techniques. Firstly, we employed data augmentation to increase the diversity of the training data and improve the model's generalization ability. Secondly, we used a strong image backbone to extract more informative features from the input data. Thirdly, we incorporated a 3D unet head to better capture the spatial information of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.18140","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-28T08:03:34Z","cross_cats_sorted":[],"title_canon_sha256":"1acaa66cb1a9f80f1f49a9e74b6dcfd1a916f648f90d962016626124162ccac4","abstract_canon_sha256":"3f596069a90ee9bf6cde63fe50ded1e1e2934c92d82cb7de245d5c10498d796d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:12.851560Z","signature_b64":"Rtt0gEq+v/zTBGeyvEyUoPU6iZRrB4N8irnJCPg5Cghb+o03Dev7h2xQ1e/URPNicxGYViPwA6wfIdCZRuvBCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00f069f7aca968ee4045fea82f19f7c20bfd89c5e1ef3ca48e247b1c76b69175","last_reissued_at":"2026-07-05T07:50:12.851131Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:12.851131Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OccTransformer: Improving BEVFormer for 3D camera-only occupancy prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Borun Xu, Chuixin Kong, Donglai Wei, Jian Liu, Ruibo Ming, Sipeng Zhang, Wenyuan Zhang, Xianming Liu, Yikang Ding, Yuhang Wu","submitted_at":"2024-02-28T08:03:34Z","abstract_excerpt":"This technical report presents our solution, \"occTransformer\" for the 3D occupancy prediction track in the autonomous driving challenge at CVPR 2023. Our method builds upon the strong baseline BEVFormer and improves its performance through several simple yet effective techniques. Firstly, we employed data augmentation to increase the diversity of the training data and improve the model's generalization ability. Secondly, we used a strong image backbone to extract more informative features from the input data. Thirdly, we incorporated a 3D unet head to better capture the spatial information of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.18140","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.18140/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.18140","created_at":"2026-07-05T07:50:12.851193+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.18140v1","created_at":"2026-07-05T07:50:12.851193+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.18140","created_at":"2026-07-05T07:50:12.851193+00:00"},{"alias_kind":"pith_short_12","alias_value":"ADYGT55MVFUO","created_at":"2026-07-05T07:50:12.851193+00:00"},{"alias_kind":"pith_short_16","alias_value":"ADYGT55MVFUO4QCF","created_at":"2026-07-05T07:50:12.851193+00:00"},{"alias_kind":"pith_short_8","alias_value":"ADYGT55M","created_at":"2026-07-05T07:50:12.851193+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.14169","citing_title":"Spatiotemporal Decoupling for Efficient Vision-Based Occupancy Forecasting","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI","json":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI.json","graph_json":"https://pith.science/api/pith-number/ADYGT55MVFUO4QCF72UC6GPXYI/graph.json","events_json":"https://pith.science/api/pith-number/ADYGT55MVFUO4QCF72UC6GPXYI/events.json","paper":"https://pith.science/paper/ADYGT55M"},"agent_actions":{"view_html":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI","download_json":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI.json","view_paper":"https://pith.science/paper/ADYGT55M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.18140&json=true","fetch_graph":"https://pith.science/api/pith-number/ADYGT55MVFUO4QCF72UC6GPXYI/graph.json","fetch_events":"https://pith.science/api/pith-number/ADYGT55MVFUO4QCF72UC6GPXYI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI/action/storage_attestation","attest_author":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI/action/author_attestation","sign_citation":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI/action/citation_signature","submit_replication":"https://pith.science/pith/ADYGT55MVFUO4QCF72UC6GPXYI/action/replication_record"}},"created_at":"2026-07-05T07:50:12.851193+00:00","updated_at":"2026-07-05T07:50:12.851193+00:00"}