{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MKEHJDR5AWAOUEK5N2RKBNZGNA","short_pith_number":"pith:MKEHJDR5","schema_version":"1.0","canonical_sha256":"6288748e3d0580ea115d6ea2a0b72668099330c7b7a9d459530a5e06181ee1e2","source":{"kind":"arxiv","id":"2405.14200","version":2},"attestation_state":"computed","paper":{"title":"Awesome Multi-modal Object Tracking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chunhui Zhang, Hao Wen, Li Liu, Xi Zhou, Yanfeng Wang","submitted_at":"2024-05-23T05:58:10Z","abstract_excerpt":"Multi-modal object tracking (MMOT) is an emerging field that combines data from various modalities, \\eg vision (RGB), depth, thermal infrared, event, language and audio, to estimate the state of an arbitrary object in a video sequence. It is of great significance for many applications such as autonomous driving and intelligent surveillance. In recent years, MMOT has received more and more attention. However, existing MMOT algorithms mainly focus on two modalities (\\eg RGB+depth, RGB+thermal infrared, and RGB+language). To leverage more modalities, some recent efforts have been made to learn a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14200","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-23T05:58:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bdf32a40a421849abcfb45077384a4060f7bebfc9d02cd23c7a2dfb11bac1f7d","abstract_canon_sha256":"8c0240bb382defbcb15906753e0905ed0ed62cc2f80afcf8670036c30f5fa3ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:34.487992Z","signature_b64":"dPkfwK1U0MENpZkP85wY8CuaHBa8s9KAB6SRJWY42MhtqUBDm3IEnONU17jI6smnE49EReKQvxvso4WQK1aFAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6288748e3d0580ea115d6ea2a0b72668099330c7b7a9d459530a5e06181ee1e2","last_reissued_at":"2026-07-05T08:25:34.487567Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:34.487567Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Awesome Multi-modal Object Tracking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chunhui Zhang, Hao Wen, Li Liu, Xi Zhou, Yanfeng Wang","submitted_at":"2024-05-23T05:58:10Z","abstract_excerpt":"Multi-modal object tracking (MMOT) is an emerging field that combines data from various modalities, \\eg vision (RGB), depth, thermal infrared, event, language and audio, to estimate the state of an arbitrary object in a video sequence. It is of great significance for many applications such as autonomous driving and intelligent surveillance. In recent years, MMOT has received more and more attention. However, existing MMOT algorithms mainly focus on two modalities (\\eg RGB+depth, RGB+thermal infrared, and RGB+language). To leverage more modalities, some recent efforts have been made to learn a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14200","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14200/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14200","created_at":"2026-07-05T08:25:34.487624+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14200v2","created_at":"2026-07-05T08:25:34.487624+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14200","created_at":"2026-07-05T08:25:34.487624+00:00"},{"alias_kind":"pith_short_12","alias_value":"MKEHJDR5AWAO","created_at":"2026-07-05T08:25:34.487624+00:00"},{"alias_kind":"pith_short_16","alias_value":"MKEHJDR5AWAOUEK5","created_at":"2026-07-05T08:25:34.487624+00:00"},{"alias_kind":"pith_short_8","alias_value":"MKEHJDR5","created_at":"2026-07-05T08:25:34.487624+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.23783","citing_title":"Mamba-FETrack V2: Revisiting State Space Model for Frame-Event based Visual Object Tracking","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA","json":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA.json","graph_json":"https://pith.science/api/pith-number/MKEHJDR5AWAOUEK5N2RKBNZGNA/graph.json","events_json":"https://pith.science/api/pith-number/MKEHJDR5AWAOUEK5N2RKBNZGNA/events.json","paper":"https://pith.science/paper/MKEHJDR5"},"agent_actions":{"view_html":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA","download_json":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA.json","view_paper":"https://pith.science/paper/MKEHJDR5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14200&json=true","fetch_graph":"https://pith.science/api/pith-number/MKEHJDR5AWAOUEK5N2RKBNZGNA/graph.json","fetch_events":"https://pith.science/api/pith-number/MKEHJDR5AWAOUEK5N2RKBNZGNA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA/action/storage_attestation","attest_author":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA/action/author_attestation","sign_citation":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA/action/citation_signature","submit_replication":"https://pith.science/pith/MKEHJDR5AWAOUEK5N2RKBNZGNA/action/replication_record"}},"created_at":"2026-07-05T08:25:34.487624+00:00","updated_at":"2026-07-05T08:25:34.487624+00:00"}