{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CITF3GRRO3AACI7TQWGNCS3W5G","short_pith_number":"pith:CITF3GRR","schema_version":"1.0","canonical_sha256":"12265d9a3176c00123f3858cd14b76e988db991ba1dd27cce7ba546d48f5c730","source":{"kind":"arxiv","id":"2407.04928","version":1},"attestation_state":"computed","paper":{"title":"CLIPVQA:Video Quality Assessment via CLIP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.IV"],"primary_cat":"cs.CV","authors_text":"Fengchuang Xing, Guopu Zhu, Mingjie Li, Xiaochun Cao, Yuan-Gen Wang","submitted_at":"2024-07-06T02:32:28Z","abstract_excerpt":"In learning vision-language representations from web-scale data, the contrastive language-image pre-training (CLIP) mechanism has demonstrated a remarkable performance in many vision tasks. However, its application to the widely studied video quality assessment (VQA) task is still an open issue. In this paper, we propose an efficient and effective CLIP-based Transformer method for the VQA problem (CLIPVQA). Specifically, we first design an effective video frame perception paradigm with the goal of extracting the rich spatiotemporal quality and content information among video frames. Then, the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.04928","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-06T02:32:28Z","cross_cats_sorted":["eess.IV"],"title_canon_sha256":"28f29f35a49dda287569b6dd94ab4cae04006c22cc9c389c511da70f14714797","abstract_canon_sha256":"7049800c6b1d014294093e7644b32c5e0a99cfc88bef11d324b9939ac9b5db81"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:43.889241Z","signature_b64":"4DgUm5RR+VPyVu3+0phLjBwujOKt6MFUUC6b51hvEzzDgaq/8uIfYuze3HYsCqQ923T+xJYO0YkUDgQeqbgHDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"12265d9a3176c00123f3858cd14b76e988db991ba1dd27cce7ba546d48f5c730","last_reissued_at":"2026-07-05T08:40:43.888864Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:43.888864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIPVQA:Video Quality Assessment via CLIP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.IV"],"primary_cat":"cs.CV","authors_text":"Fengchuang Xing, Guopu Zhu, Mingjie Li, Xiaochun Cao, Yuan-Gen Wang","submitted_at":"2024-07-06T02:32:28Z","abstract_excerpt":"In learning vision-language representations from web-scale data, the contrastive language-image pre-training (CLIP) mechanism has demonstrated a remarkable performance in many vision tasks. However, its application to the widely studied video quality assessment (VQA) task is still an open issue. In this paper, we propose an efficient and effective CLIP-based Transformer method for the VQA problem (CLIPVQA). Specifically, we first design an effective video frame perception paradigm with the goal of extracting the rich spatiotemporal quality and content information among video frames. Then, the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.04928","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.04928/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.04928","created_at":"2026-07-05T08:40:43.888931+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.04928v1","created_at":"2026-07-05T08:40:43.888931+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.04928","created_at":"2026-07-05T08:40:43.888931+00:00"},{"alias_kind":"pith_short_12","alias_value":"CITF3GRRO3AA","created_at":"2026-07-05T08:40:43.888931+00:00"},{"alias_kind":"pith_short_16","alias_value":"CITF3GRRO3AACI7T","created_at":"2026-07-05T08:40:43.888931+00:00"},{"alias_kind":"pith_short_8","alias_value":"CITF3GRR","created_at":"2026-07-05T08:40:43.888931+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.05924","citing_title":"Multi-Branch Collaborative Learning Network for Video Quality Assessment in Industrial Video Search","ref_index":52,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G","json":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G.json","graph_json":"https://pith.science/api/pith-number/CITF3GRRO3AACI7TQWGNCS3W5G/graph.json","events_json":"https://pith.science/api/pith-number/CITF3GRRO3AACI7TQWGNCS3W5G/events.json","paper":"https://pith.science/paper/CITF3GRR"},"agent_actions":{"view_html":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G","download_json":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G.json","view_paper":"https://pith.science/paper/CITF3GRR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.04928&json=true","fetch_graph":"https://pith.science/api/pith-number/CITF3GRRO3AACI7TQWGNCS3W5G/graph.json","fetch_events":"https://pith.science/api/pith-number/CITF3GRRO3AACI7TQWGNCS3W5G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G/action/storage_attestation","attest_author":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G/action/author_attestation","sign_citation":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G/action/citation_signature","submit_replication":"https://pith.science/pith/CITF3GRRO3AACI7TQWGNCS3W5G/action/replication_record"}},"created_at":"2026-07-05T08:40:43.888931+00:00","updated_at":"2026-07-05T08:40:43.888931+00:00"}