{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JIR63VE37ELIOWRAKCIBFHPAWG","short_pith_number":"pith:JIR63VE3","schema_version":"1.0","canonical_sha256":"4a23edd49bf916875a205090129de0b1a6a3dbe023f3bf14dd45c2121f856d5d","source":{"kind":"arxiv","id":"2508.05221","version":1},"attestation_state":"computed","paper":{"title":"ReasoningTrack: Chain-of-Thought Reasoning for Long-term Vision-Language Tracking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo Jiang, Lan Chen, Liye Jin, Shiao Wang, Xiao Wang, Xufeng Lou, Zhipeng Zhang","submitted_at":"2025-08-07T10:02:07Z","abstract_excerpt":"Vision-language tracking has received increasing attention in recent years, as textual information can effectively address the inflexibility and inaccuracy associated with specifying the target object to be tracked. Existing works either directly fuse the fixed language with vision features or simply modify using attention, however, their performance is still limited. Recently, some researchers have explored using text generation to adapt to the variations in the target during tracking, however, these works fail to provide insights into the model's reasoning process and do not fully leverage t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.05221","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-07T10:02:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3c48c202bd7d49045920e9d835cb15bf6ea10a1dcd5d54792520e806535f49ea","abstract_canon_sha256":"0ac34a421ae67ac1b93fe90335e455518492e242fec3b0a195d741c2018e462b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:50:09.124200Z","signature_b64":"f5atv8Nkq8eI5sZ4LIypvzGsFMJwR85zzt1JkengM1rQGJX3ckOzxL/skDNUp+4dgnAdgjTSkUsfziCkNHphDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a23edd49bf916875a205090129de0b1a6a3dbe023f3bf14dd45c2121f856d5d","last_reissued_at":"2026-07-05T11:50:09.123773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:50:09.123773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReasoningTrack: Chain-of-Thought Reasoning for Long-term Vision-Language Tracking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bo Jiang, Lan Chen, Liye Jin, Shiao Wang, Xiao Wang, Xufeng Lou, Zhipeng Zhang","submitted_at":"2025-08-07T10:02:07Z","abstract_excerpt":"Vision-language tracking has received increasing attention in recent years, as textual information can effectively address the inflexibility and inaccuracy associated with specifying the target object to be tracked. Existing works either directly fuse the fixed language with vision features or simply modify using attention, however, their performance is still limited. Recently, some researchers have explored using text generation to adapt to the variations in the target during tracking, however, these works fail to provide insights into the model's reasoning process and do not fully leverage t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.05221","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.05221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.05221","created_at":"2026-07-05T11:50:09.123829+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.05221v1","created_at":"2026-07-05T11:50:09.123829+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.05221","created_at":"2026-07-05T11:50:09.123829+00:00"},{"alias_kind":"pith_short_12","alias_value":"JIR63VE37ELI","created_at":"2026-07-05T11:50:09.123829+00:00"},{"alias_kind":"pith_short_16","alias_value":"JIR63VE37ELIOWRA","created_at":"2026-07-05T11:50:09.123829+00:00"},{"alias_kind":"pith_short_8","alias_value":"JIR63VE3","created_at":"2026-07-05T11:50:09.123829+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29357","citing_title":"Dynamic Parsing and Updating Natural Language Specification using VLMs for Robust Vision-Language Tracking","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.22799","citing_title":"VPTracker: Global Vision-Language Tracking via Visual Prompt","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10611","citing_title":"Molmo2: Open Weights and Data for Vision-Language Models with Video Understanding and Grounding","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10527","citing_title":"STORM: End-to-End Referring Multi-Object Tracking in Videos","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08014","citing_title":"Bridging Time and Space: Decoupled Spatio-Temporal Alignment for Video Grounding","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG","json":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG.json","graph_json":"https://pith.science/api/pith-number/JIR63VE37ELIOWRAKCIBFHPAWG/graph.json","events_json":"https://pith.science/api/pith-number/JIR63VE37ELIOWRAKCIBFHPAWG/events.json","paper":"https://pith.science/paper/JIR63VE3"},"agent_actions":{"view_html":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG","download_json":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG.json","view_paper":"https://pith.science/paper/JIR63VE3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.05221&json=true","fetch_graph":"https://pith.science/api/pith-number/JIR63VE37ELIOWRAKCIBFHPAWG/graph.json","fetch_events":"https://pith.science/api/pith-number/JIR63VE37ELIOWRAKCIBFHPAWG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG/action/storage_attestation","attest_author":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG/action/author_attestation","sign_citation":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG/action/citation_signature","submit_replication":"https://pith.science/pith/JIR63VE37ELIOWRAKCIBFHPAWG/action/replication_record"}},"created_at":"2026-07-05T11:50:09.123829+00:00","updated_at":"2026-07-05T11:50:09.123829+00:00"}