{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6N37WMPZO5Y7DD52OK5UUI5535","short_pith_number":"pith:6N37WMPZ","schema_version":"1.0","canonical_sha256":"f377fb31f97771f18fba72bb4a23bddf655bc108c1131b07b2950ebbb2bb48a4","source":{"kind":"arxiv","id":"2505.23189","version":1},"attestation_state":"computed","paper":{"title":"TrackVLA: Embodied Visual Tracking in the Wild","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Anqi Li, Fangwei Zhong, He Wang, Jiahang Liu, Jiazhao Zhang, Junzhi Yu, Kui Wu, Minghan Li, Shaoan Wang, Zhizheng Zhang","submitted_at":"2025-05-29T07:28:09Z","abstract_excerpt":"Embodied visual tracking is a fundamental skill in Embodied AI, enabling an agent to follow a specific target in dynamic environments using only egocentric vision. This task is inherently challenging as it requires both accurate target recognition and effective trajectory planning under conditions of severe occlusion and high scene dynamics. Existing approaches typically address this challenge through a modular separation of recognition and planning. In this work, we propose TrackVLA, a Vision-Language-Action (VLA) model that learns the synergy between object recognition and trajectory plannin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23189","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-05-29T07:28:09Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"c053e69f77825908039d6824da66683916386e6cf898beb317b2fb264f5bb94c","abstract_canon_sha256":"c242a1280b7a9ce1877e72f90771054031a2574fa2a72205da113d030152200d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:57.317820Z","signature_b64":"ztizNiwPV5fZmlzSx+FdU8MSjPtC6aObqF60vBD9OC2SqHh9yOXTQTyatYrLliAWMc04AMUaN3IuwsOUGuxPDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f377fb31f97771f18fba72bb4a23bddf655bc108c1131b07b2950ebbb2bb48a4","last_reissued_at":"2026-07-05T11:11:57.317328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:57.317328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TrackVLA: Embodied Visual Tracking in the Wild","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Anqi Li, Fangwei Zhong, He Wang, Jiahang Liu, Jiazhao Zhang, Junzhi Yu, Kui Wu, Minghan Li, Shaoan Wang, Zhizheng Zhang","submitted_at":"2025-05-29T07:28:09Z","abstract_excerpt":"Embodied visual tracking is a fundamental skill in Embodied AI, enabling an agent to follow a specific target in dynamic environments using only egocentric vision. This task is inherently challenging as it requires both accurate target recognition and effective trajectory planning under conditions of severe occlusion and high scene dynamics. Existing approaches typically address this challenge through a modular separation of recognition and planning. In this work, we propose TrackVLA, a Vision-Language-Action (VLA) model that learns the synergy between object recognition and trajectory plannin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23189","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23189/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23189","created_at":"2026-07-05T11:11:57.317385+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23189v1","created_at":"2026-07-05T11:11:57.317385+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23189","created_at":"2026-07-05T11:11:57.317385+00:00"},{"alias_kind":"pith_short_12","alias_value":"6N37WMPZO5Y7","created_at":"2026-07-05T11:11:57.317385+00:00"},{"alias_kind":"pith_short_16","alias_value":"6N37WMPZO5Y7DD52","created_at":"2026-07-05T11:11:57.317385+00:00"},{"alias_kind":"pith_short_8","alias_value":"6N37WMPZ","created_at":"2026-07-05T11:11:57.317385+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25880","citing_title":"USS: Unified Spatial-Semantic Prompts for Embodied Visual Tracking with Latent Dynamics Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18634","citing_title":"EffiNav: Fusing Depth and Vision-Language for Efficient Object Goal Navigation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03682","citing_title":"GN0: Toward a Unified Paradigm for Generation, Evaluation, and Policy Learning in Visual-Language Navigation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01848","citing_title":"RescueBench: Can Embodied Agents Save Lives in the Wild ?","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21714","citing_title":"AstraNav-World: World Model for Foresight Control and Consistency","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09005","citing_title":"Towards Backdoor-Based Ownership Verification for Vision-Language-Action Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09441","citing_title":"Beyond Isolation: A Unified Benchmark for General-Purpose Navigation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24086","citing_title":"AsyncShield: A Plug-and-Play Edge Adapter for Asynchronous Cloud-based VLA Navigation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21453","citing_title":"Instance-level Visual Active Tracking with Occlusion-Aware Planning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20347","citing_title":"A Vision-Language-Action Model for Adaptive Ultrasound-Guided Needle Insertion and Needle Tracking","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13654","citing_title":"Vision-and-Language Navigation for UAVs: Progress, Challenges, and a Research Roadmap","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535","json":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535.json","graph_json":"https://pith.science/api/pith-number/6N37WMPZO5Y7DD52OK5UUI5535/graph.json","events_json":"https://pith.science/api/pith-number/6N37WMPZO5Y7DD52OK5UUI5535/events.json","paper":"https://pith.science/paper/6N37WMPZ"},"agent_actions":{"view_html":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535","download_json":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535.json","view_paper":"https://pith.science/paper/6N37WMPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23189&json=true","fetch_graph":"https://pith.science/api/pith-number/6N37WMPZO5Y7DD52OK5UUI5535/graph.json","fetch_events":"https://pith.science/api/pith-number/6N37WMPZO5Y7DD52OK5UUI5535/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535/action/storage_attestation","attest_author":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535/action/author_attestation","sign_citation":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535/action/citation_signature","submit_replication":"https://pith.science/pith/6N37WMPZO5Y7DD52OK5UUI5535/action/replication_record"}},"created_at":"2026-07-05T11:11:57.317385+00:00","updated_at":"2026-07-05T11:11:57.317385+00:00"}