{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:A6CR5O5RX2TRNFYG66EZFCAAKX","short_pith_number":"pith:A6CR5O5R","schema_version":"1.0","canonical_sha256":"07851ebbb1bea7169706f78992880055ebf104061814743a399e30e689b3c5e3","source":{"kind":"arxiv","id":"2503.12769","version":1},"attestation_state":"computed","paper":{"title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jian-Fang Hu, Kun-Yu Lin, Qize Yang, Shenghao Fu, Wei-Shi Zheng, Xiaohua Xie, Xihan Wei, Yi-Xing Peng, Yuan-Ming Li","submitted_at":"2025-03-17T03:05:31Z","abstract_excerpt":"Recent advances in Large Multi-modal Models (LMMs) are primarily focused on offline video understanding. Instead, streaming video understanding poses great challenges to recent models due to its time-sensitive, omni-modal and interactive characteristics. In this work, we aim to extend the streaming video understanding from a new perspective and propose a novel task named Visual Instruction Feedback in which models should be aware of visual contents and learn to extract instructions from them. For example, when users wave their hands to agents, agents should recognize the gesture and start conv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.12769","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-17T03:05:31Z","cross_cats_sorted":[],"title_canon_sha256":"748b9e0f4b710c92e95df73f884590d23e8e997976478f978681d70590dfbb88","abstract_canon_sha256":"d74e4eb7125e7b8b47053902f7a21c620dbb7392320e14e64c2fed90c86e3c30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:32:45.760957Z","signature_b64":"sxEC3BUY59KhXKlkhmiWKiuOyphdW97y4Q8HvlR/mCqfOjb6mLvwXimjriYqDzaliJzeuoDoNbC/wWA2a1i+BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07851ebbb1bea7169706f78992880055ebf104061814743a399e30e689b3c5e3","last_reissued_at":"2026-07-05T10:32:45.760428Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:32:45.760428Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jian-Fang Hu, Kun-Yu Lin, Qize Yang, Shenghao Fu, Wei-Shi Zheng, Xiaohua Xie, Xihan Wei, Yi-Xing Peng, Yuan-Ming Li","submitted_at":"2025-03-17T03:05:31Z","abstract_excerpt":"Recent advances in Large Multi-modal Models (LMMs) are primarily focused on offline video understanding. Instead, streaming video understanding poses great challenges to recent models due to its time-sensitive, omni-modal and interactive characteristics. In this work, we aim to extend the streaming video understanding from a new perspective and propose a novel task named Visual Instruction Feedback in which models should be aware of visual contents and learn to extract instructions from them. For example, when users wave their hands to agents, agents should recognize the gesture and start conv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.12769","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.12769/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.12769","created_at":"2026-07-05T10:32:45.760500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.12769v1","created_at":"2026-07-05T10:32:45.760500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.12769","created_at":"2026-07-05T10:32:45.760500+00:00"},{"alias_kind":"pith_short_12","alias_value":"A6CR5O5RX2TR","created_at":"2026-07-05T10:32:45.760500+00:00"},{"alias_kind":"pith_short_16","alias_value":"A6CR5O5RX2TRNFYG","created_at":"2026-07-05T10:32:45.760500+00:00"},{"alias_kind":"pith_short_8","alias_value":"A6CR5O5R","created_at":"2026-07-05T10:32:45.760500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00248","citing_title":"Seed2.0 Model Card: Towards Intelligence Frontier for Real-World Complexity","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06991","citing_title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01707","citing_title":"StreamGaze: Gaze-Guided Temporal Reasoning and Proactive Understanding in Streaming Videos","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20633","citing_title":"Seed1.8 Model Card: Towards Generalized Real-World Agency","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24317","citing_title":"Don't Pause! Every prediction matters in a streaming video","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01858","citing_title":"Decouple and Cache: KV Cache Construction for Streaming Video Understanding","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX","json":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX.json","graph_json":"https://pith.science/api/pith-number/A6CR5O5RX2TRNFYG66EZFCAAKX/graph.json","events_json":"https://pith.science/api/pith-number/A6CR5O5RX2TRNFYG66EZFCAAKX/events.json","paper":"https://pith.science/paper/A6CR5O5R"},"agent_actions":{"view_html":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX","download_json":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX.json","view_paper":"https://pith.science/paper/A6CR5O5R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.12769&json=true","fetch_graph":"https://pith.science/api/pith-number/A6CR5O5RX2TRNFYG66EZFCAAKX/graph.json","fetch_events":"https://pith.science/api/pith-number/A6CR5O5RX2TRNFYG66EZFCAAKX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX/action/storage_attestation","attest_author":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX/action/author_attestation","sign_citation":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX/action/citation_signature","submit_replication":"https://pith.science/pith/A6CR5O5RX2TRNFYG66EZFCAAKX/action/replication_record"}},"created_at":"2026-07-05T10:32:45.760500+00:00","updated_at":"2026-07-05T10:32:45.760500+00:00"}