{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S5CIFK3QEK6JZFEZ4SFS5ZKWP4","short_pith_number":"pith:S5CIFK3Q","schema_version":"1.0","canonical_sha256":"974482ab7022bc9c9499e48b2ee5567f32594e0eb16e85aead18790d2c8bf03d","source":{"kind":"arxiv","id":"2411.12293","version":1},"attestation_state":"computed","paper":{"title":"Generative Timelines for Instructed Visual Assembly","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.HC","cs.MM"],"primary_cat":"cs.CV","authors_text":"Alejandro Pardo, Bernard Ghanem, Bryan Russell, Fabian Caba Heilbron, Josef Sivic, Jui-Hsien Wang","submitted_at":"2024-11-19T07:26:30Z","abstract_excerpt":"The objective of this work is to manipulate visual timelines (e.g. a video) through natural language instructions, making complex timeline editing tasks accessible to non-expert or potentially even disabled users. We call this task Instructed visual assembly. This task is challenging as it requires (i) identifying relevant visual content in the input timeline as well as retrieving relevant visual content in a given input (video) collection, (ii) understanding the input natural language instruction, and (iii) performing the desired edits of the input visual timeline to produce an output timelin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.12293","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-19T07:26:30Z","cross_cats_sorted":["cs.HC","cs.MM"],"title_canon_sha256":"b1fe7239c0d954c84475631b692ca2e76d663640a3521d13bbedf2a2b2075920","abstract_canon_sha256":"2fcb70792c881d6da8013e922442aa0c0f296d7bc75d77b1d9a0927bbf2e8cb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:37:18.974200Z","signature_b64":"2NlGZrSq1s2NSSqbIG2CofsI+Mxs9TxNe2npdwrchWghUJs3R6baNCbRMru1CC/AvFQNzzThTZmq419sX5gJAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"974482ab7022bc9c9499e48b2ee5567f32594e0eb16e85aead18790d2c8bf03d","last_reissued_at":"2026-07-05T09:37:18.973724Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:37:18.973724Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generative Timelines for Instructed Visual Assembly","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.HC","cs.MM"],"primary_cat":"cs.CV","authors_text":"Alejandro Pardo, Bernard Ghanem, Bryan Russell, Fabian Caba Heilbron, Josef Sivic, Jui-Hsien Wang","submitted_at":"2024-11-19T07:26:30Z","abstract_excerpt":"The objective of this work is to manipulate visual timelines (e.g. a video) through natural language instructions, making complex timeline editing tasks accessible to non-expert or potentially even disabled users. We call this task Instructed visual assembly. This task is challenging as it requires (i) identifying relevant visual content in the input timeline as well as retrieving relevant visual content in a given input (video) collection, (ii) understanding the input natural language instruction, and (iii) performing the desired edits of the input visual timeline to produce an output timelin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.12293","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.12293/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.12293","created_at":"2026-07-05T09:37:18.973780+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.12293v1","created_at":"2026-07-05T09:37:18.973780+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.12293","created_at":"2026-07-05T09:37:18.973780+00:00"},{"alias_kind":"pith_short_12","alias_value":"S5CIFK3QEK6J","created_at":"2026-07-05T09:37:18.973780+00:00"},{"alias_kind":"pith_short_16","alias_value":"S5CIFK3QEK6JZFEZ","created_at":"2026-07-05T09:37:18.973780+00:00"},{"alias_kind":"pith_short_8","alias_value":"S5CIFK3Q","created_at":"2026-07-05T09:37:18.973780+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.10456","citing_title":"A Benchmark and Multi-Agent System for Instruction-driven Cinematic Video Compilation","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4","json":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4.json","graph_json":"https://pith.science/api/pith-number/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/graph.json","events_json":"https://pith.science/api/pith-number/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/events.json","paper":"https://pith.science/paper/S5CIFK3Q"},"agent_actions":{"view_html":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4","download_json":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4.json","view_paper":"https://pith.science/paper/S5CIFK3Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.12293&json=true","fetch_graph":"https://pith.science/api/pith-number/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/graph.json","fetch_events":"https://pith.science/api/pith-number/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/action/storage_attestation","attest_author":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/action/author_attestation","sign_citation":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/action/citation_signature","submit_replication":"https://pith.science/pith/S5CIFK3QEK6JZFEZ4SFS5ZKWP4/action/replication_record"}},"created_at":"2026-07-05T09:37:18.973780+00:00","updated_at":"2026-07-05T09:37:18.973780+00:00"}