{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XOLJKWRPRYNHH3NM4DS5XVPS33","short_pith_number":"pith:XOLJKWRP","schema_version":"1.0","canonical_sha256":"bb96955a2f8e1a73edace0e5dbd5f2deefc012353d0a25e785d00f2f79cd4060","source":{"kind":"arxiv","id":"2310.19060","version":1},"attestation_state":"computed","paper":{"title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Lu Hou, Shicheng Li, Shuhuai Ren, Sishuo Chen, Xu Sun","submitted_at":"2023-10-29T16:25:32Z","abstract_excerpt":"Large-scale video-language pre-training has made remarkable strides in advancing video-language understanding tasks. However, the heavy computational burden of video encoding remains a formidable efficiency bottleneck, particularly for long-form videos. These videos contain massive visual tokens due to their inherent 3D properties and spatiotemporal redundancy, making it challenging to capture complex temporal and spatial relationships. To tackle this issue, we propose an efficient method called TEmporal-Spatial Token Aggregation (TESTA). TESTA condenses video semantics by adaptively aggregati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.19060","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-29T16:25:32Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a1758bd11fecf6c5c261dfa6417e83916c0cb3d6150b8f683ae73a8ad7c0bde7","abstract_canon_sha256":"98388e3964b14db300e7c28b1999206d8fc39b8c74506c08fdd68ca0472ba2ec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:45.465768Z","signature_b64":"m59c2Mi4HFTePHVQMcLH+SVtM7BzOS07GOaZFliN6ANNl0YWxnv8RKQR4Z2tcQwwHI3207onfIFujRCfd5cRCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb96955a2f8e1a73edace0e5dbd5f2deefc012353d0a25e785d00f2f79cd4060","last_reissued_at":"2026-07-05T07:06:45.465251Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:45.465251Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Lu Hou, Shicheng Li, Shuhuai Ren, Sishuo Chen, Xu Sun","submitted_at":"2023-10-29T16:25:32Z","abstract_excerpt":"Large-scale video-language pre-training has made remarkable strides in advancing video-language understanding tasks. However, the heavy computational burden of video encoding remains a formidable efficiency bottleneck, particularly for long-form videos. These videos contain massive visual tokens due to their inherent 3D properties and spatiotemporal redundancy, making it challenging to capture complex temporal and spatial relationships. To tackle this issue, we propose an efficient method called TEmporal-Spatial Token Aggregation (TESTA). TESTA condenses video semantics by adaptively aggregati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19060","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19060/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.19060","created_at":"2026-07-05T07:06:45.465308+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.19060v1","created_at":"2026-07-05T07:06:45.465308+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19060","created_at":"2026-07-05T07:06:45.465308+00:00"},{"alias_kind":"pith_short_12","alias_value":"XOLJKWRPRYNH","created_at":"2026-07-05T07:06:45.465308+00:00"},{"alias_kind":"pith_short_16","alias_value":"XOLJKWRPRYNHH3NM","created_at":"2026-07-05T07:06:45.465308+00:00"},{"alias_kind":"pith_short_8","alias_value":"XOLJKWRP","created_at":"2026-07-05T07:06:45.465308+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2403.00476","citing_title":"TempCompass: Do Video LLMs Really Understand Videos?","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17434","citing_title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14149","citing_title":"One Token per Highly Selective Frame: Towards Extreme Compression for Long Video Understanding","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33","json":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33.json","graph_json":"https://pith.science/api/pith-number/XOLJKWRPRYNHH3NM4DS5XVPS33/graph.json","events_json":"https://pith.science/api/pith-number/XOLJKWRPRYNHH3NM4DS5XVPS33/events.json","paper":"https://pith.science/paper/XOLJKWRP"},"agent_actions":{"view_html":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33","download_json":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33.json","view_paper":"https://pith.science/paper/XOLJKWRP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.19060&json=true","fetch_graph":"https://pith.science/api/pith-number/XOLJKWRPRYNHH3NM4DS5XVPS33/graph.json","fetch_events":"https://pith.science/api/pith-number/XOLJKWRPRYNHH3NM4DS5XVPS33/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33/action/storage_attestation","attest_author":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33/action/author_attestation","sign_citation":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33/action/citation_signature","submit_replication":"https://pith.science/pith/XOLJKWRPRYNHH3NM4DS5XVPS33/action/replication_record"}},"created_at":"2026-07-05T07:06:45.465308+00:00","updated_at":"2026-07-05T07:06:45.465308+00:00"}