{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OYQY4VBD2XX6W4HMQWATOIN5EX","short_pith_number":"pith:OYQY4VBD","schema_version":"1.0","canonical_sha256":"76218e5423d5efeb70ec85813721bd25e14b8575e9a91ca1ed43c9f83eb397dd","source":{"kind":"arxiv","id":"2503.10781","version":3},"attestation_state":"computed","paper":{"title":"Large-scale Pre-training for Grounded Video Caption Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Evangelos Kazakos, Josef Sivic","submitted_at":"2025-03-13T18:21:07Z","abstract_excerpt":"We propose a novel approach for captioning and object grounding in video, where the objects in the caption are grounded in the video via temporally dense bounding boxes. We introduce the following contributions. First, we present a large-scale automatic annotation method that aggregates frame-level captions grounded with bounding boxes into temporally dense and consistent annotations. We apply this approach on the HowTo100M dataset to construct a large-scale pre-training dataset, named HowToGround1M. We also introduce a Grounded Video Caption Generation model, dubbed GROVE, and pre-train the m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10781","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-13T18:21:07Z","cross_cats_sorted":[],"title_canon_sha256":"83071162112298aaf985ee46a64c775af02023e17cc0d1b7f5fd20faa916ba55","abstract_canon_sha256":"c8e636a25ea9ce80c8b4ecbbf1af7c354361301345bf64938f694d177261dc3e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:07:03.961547Z","signature_b64":"UkvgXHMZSBjwly8zfA9A7hpiH5Zx5js+RTCUjOw1DFndK636dxMrWMTtHOdVtcjpTtp+6bDWTlSMzHUT2xacDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76218e5423d5efeb70ec85813721bd25e14b8575e9a91ca1ed43c9f83eb397dd","last_reissued_at":"2026-07-05T12:07:03.961148Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:07:03.961148Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large-scale Pre-training for Grounded Video Caption Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Evangelos Kazakos, Josef Sivic","submitted_at":"2025-03-13T18:21:07Z","abstract_excerpt":"We propose a novel approach for captioning and object grounding in video, where the objects in the caption are grounded in the video via temporally dense bounding boxes. We introduce the following contributions. First, we present a large-scale automatic annotation method that aggregates frame-level captions grounded with bounding boxes into temporally dense and consistent annotations. We apply this approach on the HowTo100M dataset to construct a large-scale pre-training dataset, named HowToGround1M. We also introduce a Grounded Video Caption Generation model, dubbed GROVE, and pre-train the m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10781","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10781","created_at":"2026-07-05T12:07:03.961199+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10781v3","created_at":"2026-07-05T12:07:03.961199+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10781","created_at":"2026-07-05T12:07:03.961199+00:00"},{"alias_kind":"pith_short_12","alias_value":"OYQY4VBD2XX6","created_at":"2026-07-05T12:07:03.961199+00:00"},{"alias_kind":"pith_short_16","alias_value":"OYQY4VBD2XX6W4HM","created_at":"2026-07-05T12:07:03.961199+00:00"},{"alias_kind":"pith_short_8","alias_value":"OYQY4VBD","created_at":"2026-07-05T12:07:03.961199+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.03501","citing_title":"Strefer: Empowering Video LLMs with Space-Time Referring and Reasoning via Synthetic Instruction Data","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX","json":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX.json","graph_json":"https://pith.science/api/pith-number/OYQY4VBD2XX6W4HMQWATOIN5EX/graph.json","events_json":"https://pith.science/api/pith-number/OYQY4VBD2XX6W4HMQWATOIN5EX/events.json","paper":"https://pith.science/paper/OYQY4VBD"},"agent_actions":{"view_html":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX","download_json":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX.json","view_paper":"https://pith.science/paper/OYQY4VBD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10781&json=true","fetch_graph":"https://pith.science/api/pith-number/OYQY4VBD2XX6W4HMQWATOIN5EX/graph.json","fetch_events":"https://pith.science/api/pith-number/OYQY4VBD2XX6W4HMQWATOIN5EX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX/action/storage_attestation","attest_author":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX/action/author_attestation","sign_citation":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX/action/citation_signature","submit_replication":"https://pith.science/pith/OYQY4VBD2XX6W4HMQWATOIN5EX/action/replication_record"}},"created_at":"2026-07-05T12:07:03.961199+00:00","updated_at":"2026-07-05T12:07:03.961199+00:00"}