{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SX5QEMSKQNDGC3FV4DY6ZAQSVS","short_pith_number":"pith:SX5QEMSK","schema_version":"1.0","canonical_sha256":"95fb02324a8346616cb5e0f1ec8212acbae6d1ec2201f71994e4d2a9654c6045","source":{"kind":"arxiv","id":"2502.16427","version":2},"attestation_state":"computed","paper":{"title":"Fine-Grained Captioning of Long Videos through Scene Graph Consolidation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohyung Han, Sanghyeok Chu, Seonguk Seo","submitted_at":"2025-02-23T03:59:05Z","abstract_excerpt":"Recent advances in vision-language models have led to impressive progress in caption generation for images and short video clips. However, these models remain constrained by their limited temporal receptive fields, making it difficult to produce coherent and comprehensive captions for long videos. While several methods have been proposed to aggregate information across video segments, they often rely on supervised fine-tuning or incur significant computational overhead. To address these challenges, we introduce a novel framework for long video captioning based on graph consolidation. Our appro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.16427","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-23T03:59:05Z","cross_cats_sorted":[],"title_canon_sha256":"93ea56cab7b4432dcbc0e8c8c79865213d58711f4b6217f7a09a1259adbb8c0a","abstract_canon_sha256":"d0e442a25e2fd64193dbc5363c66a0463850304e086a6b74f6c32a2c331adca4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:25.242277Z","signature_b64":"48WgDW1Y09Bb10RdFsyHwoE9RUgLv308ok3VjUL83+WYWgGFgBCe3Nwp2zQLQEZbyXWTkmsH5L77VybbXtZ/Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95fb02324a8346616cb5e0f1ec8212acbae6d1ec2201f71994e4d2a9654c6045","last_reissued_at":"2026-07-05T11:32:25.241773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:25.241773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fine-Grained Captioning of Long Videos through Scene Graph Consolidation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohyung Han, Sanghyeok Chu, Seonguk Seo","submitted_at":"2025-02-23T03:59:05Z","abstract_excerpt":"Recent advances in vision-language models have led to impressive progress in caption generation for images and short video clips. However, these models remain constrained by their limited temporal receptive fields, making it difficult to produce coherent and comprehensive captions for long videos. While several methods have been proposed to aggregate information across video segments, they often rely on supervised fine-tuning or incur significant computational overhead. To address these challenges, we introduce a novel framework for long video captioning based on graph consolidation. Our appro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.16427","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.16427/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.16427","created_at":"2026-07-05T11:32:25.241832+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.16427v2","created_at":"2026-07-05T11:32:25.241832+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.16427","created_at":"2026-07-05T11:32:25.241832+00:00"},{"alias_kind":"pith_short_12","alias_value":"SX5QEMSKQNDG","created_at":"2026-07-05T11:32:25.241832+00:00"},{"alias_kind":"pith_short_16","alias_value":"SX5QEMSKQNDGC3FV","created_at":"2026-07-05T11:32:25.241832+00:00"},{"alias_kind":"pith_short_8","alias_value":"SX5QEMSK","created_at":"2026-07-05T11:32:25.241832+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25842","citing_title":"Graph it first! Enabling Reasoning on Long-form Egocentric Videos through Scene Graphs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21949","citing_title":"CapRiCorn-1K: A Comprehensive Benchmark for Video Captioning and Subject Referential Consistency Across Temporal Scales","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25842","citing_title":"Graph it first! Enabling Reasoning on Long-form Egocentric Videos through Scene Graphs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02425","citing_title":"Learning to Evolve Scenes: Reasoning about Human Activities with Scene Graphs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":91,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS","json":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS.json","graph_json":"https://pith.science/api/pith-number/SX5QEMSKQNDGC3FV4DY6ZAQSVS/graph.json","events_json":"https://pith.science/api/pith-number/SX5QEMSKQNDGC3FV4DY6ZAQSVS/events.json","paper":"https://pith.science/paper/SX5QEMSK"},"agent_actions":{"view_html":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS","download_json":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS.json","view_paper":"https://pith.science/paper/SX5QEMSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.16427&json=true","fetch_graph":"https://pith.science/api/pith-number/SX5QEMSKQNDGC3FV4DY6ZAQSVS/graph.json","fetch_events":"https://pith.science/api/pith-number/SX5QEMSKQNDGC3FV4DY6ZAQSVS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS/action/storage_attestation","attest_author":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS/action/author_attestation","sign_citation":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS/action/citation_signature","submit_replication":"https://pith.science/pith/SX5QEMSKQNDGC3FV4DY6ZAQSVS/action/replication_record"}},"created_at":"2026-07-05T11:32:25.241832+00:00","updated_at":"2026-07-05T11:32:25.241832+00:00"}