{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MYBVNQPFJE7QSQ2LV6OPLGVJAM","short_pith_number":"pith:MYBVNQPF","schema_version":"1.0","canonical_sha256":"660356c1e5493f09434baf9cf59aa90339cc76e133a9a140dbdaea6747757d96","source":{"kind":"arxiv","id":"2408.02336","version":2},"attestation_state":"computed","paper":{"title":"Infusing Environmental Captions for Long-Form Video Language Grounding","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Hyogun Lee, Jinwoo Choi, Mujeen Sung, Soyeon Hong","submitted_at":"2024-08-05T09:19:52Z","abstract_excerpt":"In this work, we tackle the problem of long-form video-language grounding (VLG). Given a long-form video and a natural language query, a model should temporally localize the precise moment that answers the query. Humans can easily solve VLG tasks, even with arbitrarily long videos, by discarding irrelevant moments using extensive and robust knowledge gained from experience. Unlike humans, existing VLG methods are prone to fall into superficial cues learned from small-scale datasets, even when they are within irrelevant frames. To overcome this challenge, we propose EI-VLG, a VLG method that le"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.02336","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-05T09:19:52Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"49b5fa21bd0d742db85111feb38b17d219c4f747526d1d7b12b8a84c057a93be","abstract_canon_sha256":"67cde3add723015e9fa92b2bed16a52544d4670c47ce6e818a71c47d398f1f71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:52:38.371885Z","signature_b64":"0ig7rV0Cb8mVlNW6ISG9T5uTCEjsDS2sVUTnEkXZxVuMDJop9Q2D3SC33250Jbi1q+tkre4jdI/JJjntId5tDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"660356c1e5493f09434baf9cf59aa90339cc76e133a9a140dbdaea6747757d96","last_reissued_at":"2026-07-05T08:52:38.371411Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:52:38.371411Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Infusing Environmental Captions for Long-Form Video Language Grounding","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Hyogun Lee, Jinwoo Choi, Mujeen Sung, Soyeon Hong","submitted_at":"2024-08-05T09:19:52Z","abstract_excerpt":"In this work, we tackle the problem of long-form video-language grounding (VLG). Given a long-form video and a natural language query, a model should temporally localize the precise moment that answers the query. Humans can easily solve VLG tasks, even with arbitrarily long videos, by discarding irrelevant moments using extensive and robust knowledge gained from experience. Unlike humans, existing VLG methods are prone to fall into superficial cues learned from small-scale datasets, even when they are within irrelevant frames. To overcome this challenge, we propose EI-VLG, a VLG method that le"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.02336","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.02336/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.02336","created_at":"2026-07-05T08:52:38.371475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.02336v2","created_at":"2026-07-05T08:52:38.371475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.02336","created_at":"2026-07-05T08:52:38.371475+00:00"},{"alias_kind":"pith_short_12","alias_value":"MYBVNQPFJE7Q","created_at":"2026-07-05T08:52:38.371475+00:00"},{"alias_kind":"pith_short_16","alias_value":"MYBVNQPFJE7QSQ2L","created_at":"2026-07-05T08:52:38.371475+00:00"},{"alias_kind":"pith_short_8","alias_value":"MYBVNQPF","created_at":"2026-07-05T08:52:38.371475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.10922","citing_title":"A Survey on Video Temporal Grounding with Multimodal Large Language Model","ref_index":97,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM","json":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM.json","graph_json":"https://pith.science/api/pith-number/MYBVNQPFJE7QSQ2LV6OPLGVJAM/graph.json","events_json":"https://pith.science/api/pith-number/MYBVNQPFJE7QSQ2LV6OPLGVJAM/events.json","paper":"https://pith.science/paper/MYBVNQPF"},"agent_actions":{"view_html":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM","download_json":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM.json","view_paper":"https://pith.science/paper/MYBVNQPF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.02336&json=true","fetch_graph":"https://pith.science/api/pith-number/MYBVNQPFJE7QSQ2LV6OPLGVJAM/graph.json","fetch_events":"https://pith.science/api/pith-number/MYBVNQPFJE7QSQ2LV6OPLGVJAM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM/action/storage_attestation","attest_author":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM/action/author_attestation","sign_citation":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM/action/citation_signature","submit_replication":"https://pith.science/pith/MYBVNQPFJE7QSQ2LV6OPLGVJAM/action/replication_record"}},"created_at":"2026-07-05T08:52:38.371475+00:00","updated_at":"2026-07-05T08:52:38.371475+00:00"}