{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JA2GSEIZHHVNLFOB2WW2QWEOBU","short_pith_number":"pith:JA2GSEIZ","schema_version":"1.0","canonical_sha256":"483469111939ead595c1d5ada8588e0d1e1acf4ebf2a5a3c0e4e9b93a5495a18","source":{"kind":"arxiv","id":"2403.16048","version":2},"attestation_state":"computed","paper":{"title":"Edit3K: Universal Representation Learning for Video Editing Components","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Chen, Libo Zhang, Longyin Wen, Sijie Zhu, Tiejian Luo, Xin Gu, Yufei Wang","submitted_at":"2024-03-24T07:29:04Z","abstract_excerpt":"This paper focuses on understanding the predominant video creation pipeline, i.e., compositional video editing with six main types of editing components, including video effects, animation, transition, filter, sticker, and text. In contrast to existing visual representation learning of visual materials (i.e., images/videos), we aim to learn visual representations of editing actions/components that are generally applied on raw materials. We start by proposing the first large-scale dataset for editing components of video creation, which covers about $3,094$ editing components with $618,800$ vide"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.16048","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-24T07:29:04Z","cross_cats_sorted":[],"title_canon_sha256":"cea55f0b727701367a5fa44dc4266730e6cd46db3ad4a610c313182f471b54b5","abstract_canon_sha256":"831633b62549dba445057cba14c24153d0d6dabfb4b94ac2d88f3b026e9e20d2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:22.876865Z","signature_b64":"IT1lU+RKhVMtHrUjKvrWuyLPAhBBWKdUpfP0PlgZFzYqmVCP47iA7nfp3YSCdyMs6gVc6h7A/b3xa2uPg47VAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"483469111939ead595c1d5ada8588e0d1e1acf4ebf2a5a3c0e4e9b93a5495a18","last_reissued_at":"2026-07-05T10:19:22.876342Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:22.876342Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Edit3K: Universal Representation Learning for Video Editing Components","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Chen, Libo Zhang, Longyin Wen, Sijie Zhu, Tiejian Luo, Xin Gu, Yufei Wang","submitted_at":"2024-03-24T07:29:04Z","abstract_excerpt":"This paper focuses on understanding the predominant video creation pipeline, i.e., compositional video editing with six main types of editing components, including video effects, animation, transition, filter, sticker, and text. In contrast to existing visual representation learning of visual materials (i.e., images/videos), we aim to learn visual representations of editing actions/components that are generally applied on raw materials. We start by proposing the first large-scale dataset for editing components of video creation, which covers about $3,094$ editing components with $618,800$ vide"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.16048","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.16048/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.16048","created_at":"2026-07-05T10:19:22.876404+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.16048v2","created_at":"2026-07-05T10:19:22.876404+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.16048","created_at":"2026-07-05T10:19:22.876404+00:00"},{"alias_kind":"pith_short_12","alias_value":"JA2GSEIZHHVN","created_at":"2026-07-05T10:19:22.876404+00:00"},{"alias_kind":"pith_short_16","alias_value":"JA2GSEIZHHVNLFOB","created_at":"2026-07-05T10:19:22.876404+00:00"},{"alias_kind":"pith_short_8","alias_value":"JA2GSEIZ","created_at":"2026-07-05T10:19:22.876404+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.03276","citing_title":"VEBench:Benchmarking Large Multimodal Models for Real-World Video Editing","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03276","citing_title":"VEBench:Benchmarking Large Multimodal Models for Real-World Video Editing","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU","json":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU.json","graph_json":"https://pith.science/api/pith-number/JA2GSEIZHHVNLFOB2WW2QWEOBU/graph.json","events_json":"https://pith.science/api/pith-number/JA2GSEIZHHVNLFOB2WW2QWEOBU/events.json","paper":"https://pith.science/paper/JA2GSEIZ"},"agent_actions":{"view_html":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU","download_json":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU.json","view_paper":"https://pith.science/paper/JA2GSEIZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.16048&json=true","fetch_graph":"https://pith.science/api/pith-number/JA2GSEIZHHVNLFOB2WW2QWEOBU/graph.json","fetch_events":"https://pith.science/api/pith-number/JA2GSEIZHHVNLFOB2WW2QWEOBU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU/action/storage_attestation","attest_author":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU/action/author_attestation","sign_citation":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU/action/citation_signature","submit_replication":"https://pith.science/pith/JA2GSEIZHHVNLFOB2WW2QWEOBU/action/replication_record"}},"created_at":"2026-07-05T10:19:22.876404+00:00","updated_at":"2026-07-05T10:19:22.876404+00:00"}