{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PCHDKT7SP74XQOZLI62HGQ7XCA","short_pith_number":"pith:PCHDKT7S","schema_version":"1.0","canonical_sha256":"788e354ff27ff9783b2b47b47343f7103f8a3013dcb8338534302df9e3bb98e9","source":{"kind":"arxiv","id":"2207.01814","version":3},"attestation_state":"computed","paper":{"title":"Multimodal Frame-Scoring Transformer for Video Summarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chanhee Lee, Heuiseok Lim, Jeiyoon Park, Kiho Kwoun","submitted_at":"2022-07-05T05:14:15Z","abstract_excerpt":"As the number of video content has mushroomed in recent years, automatic video summarization has come useful when we want to just peek at the content of the video. However, there are two underlying limitations in generic video summarization task. First, most previous approaches read in just visual features as input, leaving other modality features behind. Second, existing datasets for generic video summarization are relatively insufficient to train a caption generator used for extracting text information from a video and to train the multimodal feature extractors. To address these two problems"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.01814","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-07-05T05:14:15Z","cross_cats_sorted":[],"title_canon_sha256":"8553a7dafd16039a19a921ef79b35c10acd4a7c1bdca277dbf58a41622a743d7","abstract_canon_sha256":"dd3f71736f35e36fe8fdf065992ddd31bd7588d59ca38108734e06498e712d2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:34:32.164423Z","signature_b64":"8+Y7w6VXRj/gw8agZm6W7mrisqEDCUEbvTYITDMSMLycYNmYy2HhSNNCE24PhB1CVySesfEoLZzAkwaYP1ABAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"788e354ff27ff9783b2b47b47343f7103f8a3013dcb8338534302df9e3bb98e9","last_reissued_at":"2026-07-05T05:34:32.163930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:34:32.163930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Frame-Scoring Transformer for Video Summarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chanhee Lee, Heuiseok Lim, Jeiyoon Park, Kiho Kwoun","submitted_at":"2022-07-05T05:14:15Z","abstract_excerpt":"As the number of video content has mushroomed in recent years, automatic video summarization has come useful when we want to just peek at the content of the video. However, there are two underlying limitations in generic video summarization task. First, most previous approaches read in just visual features as input, leaving other modality features behind. Second, existing datasets for generic video summarization are relatively insufficient to train a caption generator used for extracting text information from a video and to train the multimodal feature extractors. To address these two problems"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.01814","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.01814/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.01814","created_at":"2026-07-05T05:34:32.163990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.01814v3","created_at":"2026-07-05T05:34:32.163990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.01814","created_at":"2026-07-05T05:34:32.163990+00:00"},{"alias_kind":"pith_short_12","alias_value":"PCHDKT7SP74X","created_at":"2026-07-05T05:34:32.163990+00:00"},{"alias_kind":"pith_short_16","alias_value":"PCHDKT7SP74XQOZL","created_at":"2026-07-05T05:34:32.163990+00:00"},{"alias_kind":"pith_short_8","alias_value":"PCHDKT7S","created_at":"2026-07-05T05:34:32.163990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.23714","citing_title":"Towards an Automated Multimodal Approach for Video Summarization: Building a Bridge Between Text, Audio and Facial Cue-Based Summarization","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA","json":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA.json","graph_json":"https://pith.science/api/pith-number/PCHDKT7SP74XQOZLI62HGQ7XCA/graph.json","events_json":"https://pith.science/api/pith-number/PCHDKT7SP74XQOZLI62HGQ7XCA/events.json","paper":"https://pith.science/paper/PCHDKT7S"},"agent_actions":{"view_html":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA","download_json":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA.json","view_paper":"https://pith.science/paper/PCHDKT7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.01814&json=true","fetch_graph":"https://pith.science/api/pith-number/PCHDKT7SP74XQOZLI62HGQ7XCA/graph.json","fetch_events":"https://pith.science/api/pith-number/PCHDKT7SP74XQOZLI62HGQ7XCA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA/action/storage_attestation","attest_author":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA/action/author_attestation","sign_citation":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA/action/citation_signature","submit_replication":"https://pith.science/pith/PCHDKT7SP74XQOZLI62HGQ7XCA/action/replication_record"}},"created_at":"2026-07-05T05:34:32.163990+00:00","updated_at":"2026-07-05T05:34:32.163990+00:00"}