{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:STKJ6KZDPLF7GU6VQSB3MGR4XR","short_pith_number":"pith:STKJ6KZD","schema_version":"1.0","canonical_sha256":"94d49f2b237acbf353d58483b61a3cbc61147c13b17c4995e358fd8fd0e5c26f","source":{"kind":"arxiv","id":"2304.14394","version":3},"attestation_state":"computed","paper":{"title":"Unified Sequence-to-Sequence Learning for Single- and Multi-Modal Visual Object Tracking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ben Kang, Dong Wang, Houwen Peng, Huchuan Lu, Jiawen Zhu, Xin Chen","submitted_at":"2023-04-27T17:56:29Z","abstract_excerpt":"In this paper, we introduce a new sequence-to-sequence learning framework for RGB-based and multi-modal object tracking. First, we present SeqTrack for RGB-based tracking. It casts visual tracking as a sequence generation task, forecasting object bounding boxes in an autoregressive manner. This differs from previous trackers, which depend on the design of intricate head networks, such as classification and regression heads. SeqTrack employs a basic encoder-decoder transformer architecture. The encoder utilizes a bidirectional transformer for feature extraction, while the decoder generates boun"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.14394","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-04-27T17:56:29Z","cross_cats_sorted":[],"title_canon_sha256":"7c4560c3977b99f49fa56a1881dc0ce17d5c49537a654af2ea6faa61b08a6ef6","abstract_canon_sha256":"e69f0c8afafce2d042bda8c27d7bba5e866ea4d2b7187a45e2a46b43803012b9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:59.236954Z","signature_b64":"3POBrpfO9fjbDxtal9Nh11j3mfRnlxRy1GpwDV9sy19Fj7XAR0EDEQiHU53KiE2VEwfveJZtp5kVdoAsDHRjAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"94d49f2b237acbf353d58483b61a3cbc61147c13b17c4995e358fd8fd0e5c26f","last_reissued_at":"2026-07-05T08:00:59.236486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:59.236486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unified Sequence-to-Sequence Learning for Single- and Multi-Modal Visual Object Tracking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ben Kang, Dong Wang, Houwen Peng, Huchuan Lu, Jiawen Zhu, Xin Chen","submitted_at":"2023-04-27T17:56:29Z","abstract_excerpt":"In this paper, we introduce a new sequence-to-sequence learning framework for RGB-based and multi-modal object tracking. First, we present SeqTrack for RGB-based tracking. It casts visual tracking as a sequence generation task, forecasting object bounding boxes in an autoregressive manner. This differs from previous trackers, which depend on the design of intricate head networks, such as classification and regression heads. SeqTrack employs a basic encoder-decoder transformer architecture. The encoder utilizes a bidirectional transformer for feature extraction, while the decoder generates boun"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.14394","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.14394/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.14394","created_at":"2026-07-05T08:00:59.236544+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.14394v3","created_at":"2026-07-05T08:00:59.236544+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.14394","created_at":"2026-07-05T08:00:59.236544+00:00"},{"alias_kind":"pith_short_12","alias_value":"STKJ6KZDPLF7","created_at":"2026-07-05T08:00:59.236544+00:00"},{"alias_kind":"pith_short_16","alias_value":"STKJ6KZDPLF7GU6V","created_at":"2026-07-05T08:00:59.236544+00:00"},{"alias_kind":"pith_short_8","alias_value":"STKJ6KZD","created_at":"2026-07-05T08:00:59.236544+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07379","citing_title":"RELO: Reinforcement Learning to Localize for Visual Object Tracking","ref_index":299,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05359","citing_title":"Group Orthogonal Low-Rank Adaptation for RGB-T Tracking","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03716","citing_title":"Unified Multimodal Visual Tracking with Dual Mixture-of-Experts","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06092","citing_title":"Boosting Self-Supervised Tracking with Contextual Prompts and Noise Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07379","citing_title":"RELO: Reinforcement Learning to Localize for Visual Object Tracking","ref_index":298,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR","json":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR.json","graph_json":"https://pith.science/api/pith-number/STKJ6KZDPLF7GU6VQSB3MGR4XR/graph.json","events_json":"https://pith.science/api/pith-number/STKJ6KZDPLF7GU6VQSB3MGR4XR/events.json","paper":"https://pith.science/paper/STKJ6KZD"},"agent_actions":{"view_html":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR","download_json":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR.json","view_paper":"https://pith.science/paper/STKJ6KZD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.14394&json=true","fetch_graph":"https://pith.science/api/pith-number/STKJ6KZDPLF7GU6VQSB3MGR4XR/graph.json","fetch_events":"https://pith.science/api/pith-number/STKJ6KZDPLF7GU6VQSB3MGR4XR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR/action/storage_attestation","attest_author":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR/action/author_attestation","sign_citation":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR/action/citation_signature","submit_replication":"https://pith.science/pith/STKJ6KZDPLF7GU6VQSB3MGR4XR/action/replication_record"}},"created_at":"2026-07-05T08:00:59.236544+00:00","updated_at":"2026-07-05T08:00:59.236544+00:00"}