{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CJZLYI3FQDAAJR4YSAJRIZXRFE","short_pith_number":"pith:CJZLYI3F","schema_version":"1.0","canonical_sha256":"1272bc236580c004c79890131466f12925af54161b67e5c85c5beb1b99ba8c81","source":{"kind":"arxiv","id":"2505.15145","version":1},"attestation_state":"computed","paper":{"title":"CineTechBench: A Benchmark for Cinematographic Technique Understanding and Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kongming Liang, Muxi Diao, Songyu Xu, Xiangxuan Shan, Xinran Wang, Xueyan Duan, Yanhua Huang, Yuxuan Zhang, Zhanyu Ma","submitted_at":"2025-05-21T06:02:39Z","abstract_excerpt":"Cinematography is a cornerstone of film production and appreciation, shaping mood, emotion, and narrative through visual elements such as camera movement, shot composition, and lighting. Despite recent progress in multimodal large language models (MLLMs) and video generation models, the capacity of current models to grasp and reproduce cinematographic techniques remains largely uncharted, hindered by the scarcity of expert-annotated data. To bridge this gap, we present CineTechBench, a pioneering benchmark founded on precise, manual annotation by seasoned cinematography experts across key cine"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15145","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-21T06:02:39Z","cross_cats_sorted":[],"title_canon_sha256":"c5a2b94134d950b71fa93fb4aa0ebf6c459fe9594d418e1299931f5381078835","abstract_canon_sha256":"2629f5803f8554d5135cc294858efe7615fec4c54948c8daf79d8f3d9a43ae39"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:41.198503Z","signature_b64":"hUg+A+vaYTfF5KkEH4ITxEqJrWmmhLU7rgo9yppzQZM+ugbk7zShsHMj3FXMuz46nS2a6Qg8hPVIOS0Vl++SAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1272bc236580c004c79890131466f12925af54161b67e5c85c5beb1b99ba8c81","last_reissued_at":"2026-07-05T11:06:41.198031Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:41.198031Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CineTechBench: A Benchmark for Cinematographic Technique Understanding and Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Kongming Liang, Muxi Diao, Songyu Xu, Xiangxuan Shan, Xinran Wang, Xueyan Duan, Yanhua Huang, Yuxuan Zhang, Zhanyu Ma","submitted_at":"2025-05-21T06:02:39Z","abstract_excerpt":"Cinematography is a cornerstone of film production and appreciation, shaping mood, emotion, and narrative through visual elements such as camera movement, shot composition, and lighting. Despite recent progress in multimodal large language models (MLLMs) and video generation models, the capacity of current models to grasp and reproduce cinematographic techniques remains largely uncharted, hindered by the scarcity of expert-annotated data. To bridge this gap, we present CineTechBench, a pioneering benchmark founded on precise, manual annotation by seasoned cinematography experts across key cine"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15145","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15145","created_at":"2026-07-05T11:06:41.198087+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15145v1","created_at":"2026-07-05T11:06:41.198087+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15145","created_at":"2026-07-05T11:06:41.198087+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJZLYI3FQDAA","created_at":"2026-07-05T11:06:41.198087+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJZLYI3FQDAAJR4Y","created_at":"2026-07-05T11:06:41.198087+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJZLYI3F","created_at":"2026-07-05T11:06:41.198087+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24636","citing_title":"CineCap: Structured Reasoning with Spatio-Temporal Anchors for Cinematographic Video Captioning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28035","citing_title":"MTAVG-Bench 2.0: Diagnosing Failure Modes of Cinematic Expressiveness in Multi-Talker Audio-Video Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03276","citing_title":"VEBench:Benchmarking Large Multimodal Models for Real-World Video Editing","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03276","citing_title":"VEBench:Benchmarking Large Multimodal Models for Real-World Video Editing","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE","json":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE.json","graph_json":"https://pith.science/api/pith-number/CJZLYI3FQDAAJR4YSAJRIZXRFE/graph.json","events_json":"https://pith.science/api/pith-number/CJZLYI3FQDAAJR4YSAJRIZXRFE/events.json","paper":"https://pith.science/paper/CJZLYI3F"},"agent_actions":{"view_html":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE","download_json":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE.json","view_paper":"https://pith.science/paper/CJZLYI3F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15145&json=true","fetch_graph":"https://pith.science/api/pith-number/CJZLYI3FQDAAJR4YSAJRIZXRFE/graph.json","fetch_events":"https://pith.science/api/pith-number/CJZLYI3FQDAAJR4YSAJRIZXRFE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE/action/storage_attestation","attest_author":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE/action/author_attestation","sign_citation":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE/action/citation_signature","submit_replication":"https://pith.science/pith/CJZLYI3FQDAAJR4YSAJRIZXRFE/action/replication_record"}},"created_at":"2026-07-05T11:06:41.198087+00:00","updated_at":"2026-07-05T11:06:41.198087+00:00"}