{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6EEI7VULJLAQDZXHOUTNOVLP4E","short_pith_number":"pith:6EEI7VUL","schema_version":"1.0","canonical_sha256":"f1088fd68b4ac101e6e77526d7556fe123f876f9f89fee18b8b2e1c8b1a9bfb2","source":{"kind":"arxiv","id":"2502.06583","version":1},"attestation_state":"computed","paper":{"title":"Adaptive Perception for Unified Visual Multi-modal Object Tracking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bineng Zhong, Jian Yang, Liangtao Shi, Qihua Liang, Xiantao Hu, Ying Tai, Zhiyi Mo","submitted_at":"2025-02-10T15:50:26Z","abstract_excerpt":"Recently, many multi-modal trackers prioritize RGB as the dominant modality, treating other modalities as auxiliary, and fine-tuning separately various multi-modal tasks. This imbalance in modality dependence limits the ability of methods to dynamically utilize complementary information from each modality in complex scenarios, making it challenging to fully perceive the advantages of multi-modal. As a result, a unified parameter model often underperforms in various multi-modal tracking tasks. To address this issue, we propose APTrack, a novel unified tracker designed for multi-modal adaptive p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06583","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-10T15:50:26Z","cross_cats_sorted":[],"title_canon_sha256":"1de192e5b76ad63fee6ca15f70c8fa680c525a9a890c4a24fdff1a6a63922d61","abstract_canon_sha256":"ff4a9e7c0f69f427159cb449a58baa9ac02acbb1f8849c0b92d20660a623cc2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:08.824439Z","signature_b64":"EyJi0kwm/kdoV1ZnNoUW8aGlJ+XEbuWiNAEs0dXNvsBf76dNr91DTzhtppGWl/Pe7Wjp+PVEaxDYjhKTtra2Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1088fd68b4ac101e6e77526d7556fe123f876f9f89fee18b8b2e1c8b1a9bfb2","last_reissued_at":"2026-07-05T10:12:08.823890Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:08.823890Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Perception for Unified Visual Multi-modal Object Tracking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bineng Zhong, Jian Yang, Liangtao Shi, Qihua Liang, Xiantao Hu, Ying Tai, Zhiyi Mo","submitted_at":"2025-02-10T15:50:26Z","abstract_excerpt":"Recently, many multi-modal trackers prioritize RGB as the dominant modality, treating other modalities as auxiliary, and fine-tuning separately various multi-modal tasks. This imbalance in modality dependence limits the ability of methods to dynamically utilize complementary information from each modality in complex scenarios, making it challenging to fully perceive the advantages of multi-modal. As a result, a unified parameter model often underperforms in various multi-modal tracking tasks. To address this issue, we propose APTrack, a novel unified tracker designed for multi-modal adaptive p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06583","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06583/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06583","created_at":"2026-07-05T10:12:08.823957+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06583v1","created_at":"2026-07-05T10:12:08.823957+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06583","created_at":"2026-07-05T10:12:08.823957+00:00"},{"alias_kind":"pith_short_12","alias_value":"6EEI7VULJLAQ","created_at":"2026-07-05T10:12:08.823957+00:00"},{"alias_kind":"pith_short_16","alias_value":"6EEI7VULJLAQDZXH","created_at":"2026-07-05T10:12:08.823957+00:00"},{"alias_kind":"pith_short_8","alias_value":"6EEI7VUL","created_at":"2026-07-05T10:12:08.823957+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16191","citing_title":"Explicit Context Reasoning with Supervision for Visual Tracking","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E","json":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E.json","graph_json":"https://pith.science/api/pith-number/6EEI7VULJLAQDZXHOUTNOVLP4E/graph.json","events_json":"https://pith.science/api/pith-number/6EEI7VULJLAQDZXHOUTNOVLP4E/events.json","paper":"https://pith.science/paper/6EEI7VUL"},"agent_actions":{"view_html":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E","download_json":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E.json","view_paper":"https://pith.science/paper/6EEI7VUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06583&json=true","fetch_graph":"https://pith.science/api/pith-number/6EEI7VULJLAQDZXHOUTNOVLP4E/graph.json","fetch_events":"https://pith.science/api/pith-number/6EEI7VULJLAQDZXHOUTNOVLP4E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E/action/storage_attestation","attest_author":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E/action/author_attestation","sign_citation":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E/action/citation_signature","submit_replication":"https://pith.science/pith/6EEI7VULJLAQDZXHOUTNOVLP4E/action/replication_record"}},"created_at":"2026-07-05T10:12:08.823957+00:00","updated_at":"2026-07-05T10:12:08.823957+00:00"}