{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:S4YLADI3CRUMDJXGA6IPRRQ545","short_pith_number":"pith:S4YLADI3","schema_version":"1.0","canonical_sha256":"9730b00d1b1468c1a6e60790f8c61de76501b21d25879bca92f4f12d1624eb83","source":{"kind":"arxiv","id":"2608.02039","version":1},"attestation_state":"computed","paper":{"title":"RSVideo: Are Your Vision-Language Models Ready for Remote Sensing Videos?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Di Wang, Fu Lin, Haonan Guo, Haoyang Chen, Hongjie Zhou, Juhua Liu, Shiqin Wang, Yong Luo","submitted_at":"2026-08-03T10:34:42Z","abstract_excerpt":"Remote-sensing videos enable real-time observation of changes in target attributes, short-term activities, and scene evolution. They record motion, actions, interactions, and scene changes that cannot be captured by isolated images. Existing models primarily target single images or discrete temporal observations spanning a long time range. However, a unified evaluation setting for assessing vision-language models on continuous remote-sensing video understanding remains lacking. We introduce RSVideo-10K, a remote-sensing video dataset comprising 10,773 instances, 1.47 million frames, and 17.02 "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.02039","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-08-03T10:34:42Z","cross_cats_sorted":[],"title_canon_sha256":"4d1f5965af49de7efc3fc81db33a7578da4c7328072929c7929f92c9bc9a1f86","abstract_canon_sha256":"6e9d100b3d8dab935e72ac5c01ce6322ba8b84f049873075e97329b51da64ad2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T02:10:38.309408Z","signature_b64":"9RPFAcR+vibk4BnoD5O/mbiziRXRrXIkft+P1cl/feqsM49d28RAB3MFL1EnBoqU+s45PGOT0uYcWYhOaV/4CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9730b00d1b1468c1a6e60790f8c61de76501b21d25879bca92f4f12d1624eb83","last_reissued_at":"2026-08-04T02:10:38.307396Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T02:10:38.307396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RSVideo: Are Your Vision-Language Models Ready for Remote Sensing Videos?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Di Wang, Fu Lin, Haonan Guo, Haoyang Chen, Hongjie Zhou, Juhua Liu, Shiqin Wang, Yong Luo","submitted_at":"2026-08-03T10:34:42Z","abstract_excerpt":"Remote-sensing videos enable real-time observation of changes in target attributes, short-term activities, and scene evolution. They record motion, actions, interactions, and scene changes that cannot be captured by isolated images. Existing models primarily target single images or discrete temporal observations spanning a long time range. However, a unified evaluation setting for assessing vision-language models on continuous remote-sensing video understanding remains lacking. We introduce RSVideo-10K, a remote-sensing video dataset comprising 10,773 instances, 1.47 million frames, and 17.02 "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.02039","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.02039/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.02039","created_at":"2026-08-04T02:10:38.309087+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.02039v1","created_at":"2026-08-04T02:10:38.309087+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.02039","created_at":"2026-08-04T02:10:38.309087+00:00"},{"alias_kind":"pith_short_12","alias_value":"S4YLADI3CRUM","created_at":"2026-08-04T02:10:38.309087+00:00"},{"alias_kind":"pith_short_16","alias_value":"S4YLADI3CRUMDJXG","created_at":"2026-08-04T02:10:38.309087+00:00"},{"alias_kind":"pith_short_8","alias_value":"S4YLADI3","created_at":"2026-08-04T02:10:38.309087+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545","json":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545.json","graph_json":"https://pith.science/api/pith-number/S4YLADI3CRUMDJXGA6IPRRQ545/graph.json","events_json":"https://pith.science/api/pith-number/S4YLADI3CRUMDJXGA6IPRRQ545/events.json","paper":"https://pith.science/paper/S4YLADI3"},"agent_actions":{"view_html":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545","download_json":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545.json","view_paper":"https://pith.science/paper/S4YLADI3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.02039&json=true","fetch_graph":"https://pith.science/api/pith-number/S4YLADI3CRUMDJXGA6IPRRQ545/graph.json","fetch_events":"https://pith.science/api/pith-number/S4YLADI3CRUMDJXGA6IPRRQ545/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545/action/storage_attestation","attest_author":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545/action/author_attestation","sign_citation":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545/action/citation_signature","submit_replication":"https://pith.science/pith/S4YLADI3CRUMDJXGA6IPRRQ545/action/replication_record"}},"created_at":"2026-08-04T02:10:38.309087+00:00","updated_at":"2026-08-04T02:10:38.309087+00:00"}