{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QCP4COD2NHULBVEQ2EB4EBQVCO","short_pith_number":"pith:QCP4COD2","schema_version":"1.0","canonical_sha256":"809fc1387a69e8b0d490d103c2061513ba6f72d71323abf95503c40a59f1bc15","source":{"kind":"arxiv","id":"2503.14498","version":1},"attestation_state":"computed","paper":{"title":"Tracking Meets Large Multimodal Models for Driving Scenario Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Ayesha Ishaq, Fahad Shahbaz Khan, Hisham Cholakkal, Jean Lahoud, Rao Muhammad Anwer, Salman Khan","submitted_at":"2025-03-18T17:59:12Z","abstract_excerpt":"Large Multimodal Models (LMMs) have recently gained prominence in autonomous driving research, showcasing promising capabilities across various emerging benchmarks. LMMs specifically designed for this domain have demonstrated effective perception, planning, and prediction skills. However, many of these methods underutilize 3D spatial and temporal elements, relying mainly on image data. As a result, their effectiveness in dynamic driving environments is limited. We propose to integrate tracking information as an additional input to recover 3D spatial and temporal details that are not effectivel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.14498","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-18T17:59:12Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"44baa4d2a75cf9c893ce8c372a1168c6e99257af442f452dd5b78bff1360e6b5","abstract_canon_sha256":"2c40562e5c2a42f5ad729a792da225652650d8535bff699c695a2c08456ee49d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:34:03.935305Z","signature_b64":"vJj8Hf0v7c3LO1OqBHwiqFkVe4QdFINIZB12EwJiiPaBXtCyXXMR718oXLlrqwXcTOwQSPsKxNT4Xmj/lSMOBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"809fc1387a69e8b0d490d103c2061513ba6f72d71323abf95503c40a59f1bc15","last_reissued_at":"2026-07-05T10:34:03.934675Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:34:03.934675Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tracking Meets Large Multimodal Models for Driving Scenario Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Ayesha Ishaq, Fahad Shahbaz Khan, Hisham Cholakkal, Jean Lahoud, Rao Muhammad Anwer, Salman Khan","submitted_at":"2025-03-18T17:59:12Z","abstract_excerpt":"Large Multimodal Models (LMMs) have recently gained prominence in autonomous driving research, showcasing promising capabilities across various emerging benchmarks. LMMs specifically designed for this domain have demonstrated effective perception, planning, and prediction skills. However, many of these methods underutilize 3D spatial and temporal elements, relying mainly on image data. As a result, their effectiveness in dynamic driving environments is limited. We propose to integrate tracking information as an additional input to recover 3D spatial and temporal details that are not effectivel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.14498","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.14498/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.14498","created_at":"2026-07-05T10:34:03.934743+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.14498v1","created_at":"2026-07-05T10:34:03.934743+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.14498","created_at":"2026-07-05T10:34:03.934743+00:00"},{"alias_kind":"pith_short_12","alias_value":"QCP4COD2NHUL","created_at":"2026-07-05T10:34:03.934743+00:00"},{"alias_kind":"pith_short_16","alias_value":"QCP4COD2NHULBVEQ","created_at":"2026-07-05T10:34:03.934743+00:00"},{"alias_kind":"pith_short_8","alias_value":"QCP4COD2","created_at":"2026-07-05T10:34:03.934743+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.17685","citing_title":"FutureSightDrive: Thinking Visually with Spatio-Temporal CoT for Autonomous Driving","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO","json":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO.json","graph_json":"https://pith.science/api/pith-number/QCP4COD2NHULBVEQ2EB4EBQVCO/graph.json","events_json":"https://pith.science/api/pith-number/QCP4COD2NHULBVEQ2EB4EBQVCO/events.json","paper":"https://pith.science/paper/QCP4COD2"},"agent_actions":{"view_html":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO","download_json":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO.json","view_paper":"https://pith.science/paper/QCP4COD2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.14498&json=true","fetch_graph":"https://pith.science/api/pith-number/QCP4COD2NHULBVEQ2EB4EBQVCO/graph.json","fetch_events":"https://pith.science/api/pith-number/QCP4COD2NHULBVEQ2EB4EBQVCO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO/action/storage_attestation","attest_author":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO/action/author_attestation","sign_citation":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO/action/citation_signature","submit_replication":"https://pith.science/pith/QCP4COD2NHULBVEQ2EB4EBQVCO/action/replication_record"}},"created_at":"2026-07-05T10:34:03.934743+00:00","updated_at":"2026-07-05T10:34:03.934743+00:00"}