{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RFUFWXAP3PZVMXMO3OGSD4MBGM","short_pith_number":"pith:RFUFWXAP","schema_version":"1.0","canonical_sha256":"89685b5c0fdbf3565d8edb8d21f1813321dad458d4139176c3d1518c1c78dad0","source":{"kind":"arxiv","id":"2502.15393","version":2},"attestation_state":"computed","paper":{"title":"LongCaptioning: Unlocking the Power of Long Video Caption Generation in Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Wen Chen, Hongchen Wei, Yaosi Hu, Zhenzhong Chen, Zhihong Tan","submitted_at":"2025-02-21T11:40:23Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated exceptional performance in video captioning tasks, particularly for short videos. However, as the length of the video increases, generating long, detailed captions becomes a significant challenge. In this paper, we investigate the limitations of LMMs in generating long captions for long videos. Our analysis reveals that open-source LMMs struggle to consistently produce outputs exceeding 300 words, leading to incomplete or overly concise descriptions of the visual content. This limitation hinders the ability of LMMs to provide comprehensive and d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.15393","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-21T11:40:23Z","cross_cats_sorted":[],"title_canon_sha256":"b901d5bd2cf82b77ba04357a98969545086384c3bd6253a9503e6d7fbbb9812b","abstract_canon_sha256":"bc0892855c9515b7b6dad124344b63adc20a4eb88f3ac5ae0331d9b9b6b0bcf8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:14.085565Z","signature_b64":"gnzqCgkHxdQrgYDat1uQwK9eZF4uLtjm9YFf6n2nP9Nvw6Ns+R0foj9X5uDoiryhCkQJh9EANmcjkxZrJ9DHDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89685b5c0fdbf3565d8edb8d21f1813321dad458d4139176c3d1518c1c78dad0","last_reissued_at":"2026-07-05T10:22:14.085019Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:14.085019Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LongCaptioning: Unlocking the Power of Long Video Caption Generation in Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Wen Chen, Hongchen Wei, Yaosi Hu, Zhenzhong Chen, Zhihong Tan","submitted_at":"2025-02-21T11:40:23Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated exceptional performance in video captioning tasks, particularly for short videos. However, as the length of the video increases, generating long, detailed captions becomes a significant challenge. In this paper, we investigate the limitations of LMMs in generating long captions for long videos. Our analysis reveals that open-source LMMs struggle to consistently produce outputs exceeding 300 words, leading to incomplete or overly concise descriptions of the visual content. This limitation hinders the ability of LMMs to provide comprehensive and d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15393","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.15393/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.15393","created_at":"2026-07-05T10:22:14.085082+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.15393v2","created_at":"2026-07-05T10:22:14.085082+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15393","created_at":"2026-07-05T10:22:14.085082+00:00"},{"alias_kind":"pith_short_12","alias_value":"RFUFWXAP3PZV","created_at":"2026-07-05T10:22:14.085082+00:00"},{"alias_kind":"pith_short_16","alias_value":"RFUFWXAP3PZVMXMO","created_at":"2026-07-05T10:22:14.085082+00:00"},{"alias_kind":"pith_short_8","alias_value":"RFUFWXAP","created_at":"2026-07-05T10:22:14.085082+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM","json":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM.json","graph_json":"https://pith.science/api/pith-number/RFUFWXAP3PZVMXMO3OGSD4MBGM/graph.json","events_json":"https://pith.science/api/pith-number/RFUFWXAP3PZVMXMO3OGSD4MBGM/events.json","paper":"https://pith.science/paper/RFUFWXAP"},"agent_actions":{"view_html":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM","download_json":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM.json","view_paper":"https://pith.science/paper/RFUFWXAP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.15393&json=true","fetch_graph":"https://pith.science/api/pith-number/RFUFWXAP3PZVMXMO3OGSD4MBGM/graph.json","fetch_events":"https://pith.science/api/pith-number/RFUFWXAP3PZVMXMO3OGSD4MBGM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM/action/storage_attestation","attest_author":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM/action/author_attestation","sign_citation":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM/action/citation_signature","submit_replication":"https://pith.science/pith/RFUFWXAP3PZVMXMO3OGSD4MBGM/action/replication_record"}},"created_at":"2026-07-05T10:22:14.085082+00:00","updated_at":"2026-07-05T10:22:14.085082+00:00"}