{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4M4MK2A6AHBHRSHU4ZLNJMHVC3","short_pith_number":"pith:4M4MK2A6","schema_version":"1.0","canonical_sha256":"e338c5681e01c278c8f4e656d4b0f516daa3330b3e6ccd7de97b337847622857","source":{"kind":"arxiv","id":"2311.08835","version":4},"attestation_state":"computed","paper":{"title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jae-Pil Heo, Sangeek Hyun, SuBeen Lee, WonJun Moon","submitted_at":"2023-11-15T10:22:35Z","abstract_excerpt":"Temporal Grounding is to identify specific moments or highlights from a video corresponding to textual descriptions. Typical approaches in temporal grounding treat all video clips equally during the encoding process regardless of their semantic relevance with the text query. Therefore, we propose Correlation-Guided DEtection TRansformer (CG-DETR), exploring to provide clues for query-associated video clips within the cross-modal attention. First, we design an adaptive cross-attention with dummy tokens. Dummy tokens conditioned by text query take portions of the attention weights, preventing ir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.08835","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-15T10:22:35Z","cross_cats_sorted":[],"title_canon_sha256":"e44462110de89884259ac28efe634156312c3d92dce18513193a205ed37ba76c","abstract_canon_sha256":"6d44d328287208391af611b9a2cc3b3124e1fce6da6bac3c69090fea0f07b2e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:51.454371Z","signature_b64":"7VEnpDEd9ZrcefGJpGv6ZBUfORVDA3TnZpZArT7S84Ss77WgU2Zd+zEn3s91j9XfmvlhKJ7VhFLTCDaCmeKbCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e338c5681e01c278c8f4e656d4b0f516daa3330b3e6ccd7de97b337847622857","last_reissued_at":"2026-07-05T08:39:51.453810Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:51.453810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jae-Pil Heo, Sangeek Hyun, SuBeen Lee, WonJun Moon","submitted_at":"2023-11-15T10:22:35Z","abstract_excerpt":"Temporal Grounding is to identify specific moments or highlights from a video corresponding to textual descriptions. Typical approaches in temporal grounding treat all video clips equally during the encoding process regardless of their semantic relevance with the text query. Therefore, we propose Correlation-Guided DEtection TRansformer (CG-DETR), exploring to provide clues for query-associated video clips within the cross-modal attention. First, we design an adaptive cross-attention with dummy tokens. Dummy tokens conditioned by text query take portions of the attention weights, preventing ir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.08835","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.08835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.08835","created_at":"2026-07-05T08:39:51.453880+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.08835v4","created_at":"2026-07-05T08:39:51.453880+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.08835","created_at":"2026-07-05T08:39:51.453880+00:00"},{"alias_kind":"pith_short_12","alias_value":"4M4MK2A6AHBH","created_at":"2026-07-05T08:39:51.453880+00:00"},{"alias_kind":"pith_short_16","alias_value":"4M4MK2A6AHBHRSHU","created_at":"2026-07-05T08:39:51.453880+00:00"},{"alias_kind":"pith_short_8","alias_value":"4M4MK2A6","created_at":"2026-07-05T08:39:51.453880+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19706","citing_title":"NEST: Narrative Event Structures in Time for Long Video Understanding","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06926","citing_title":"SVHighlights: Towards Extremely Long Sport Video Highlight Detection","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01615","citing_title":"Turing Patterns for Multimedia: Reaction-Diffusion Multi-Modal Fusion for Language-Guided Video Moment Retrieval","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01149","citing_title":"CoSTL: Comprehensive Spatial-Temporal Representation Learning for Moment Retrieval and Highlight Detection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26104","citing_title":"EVIDENT: Routing MLLM Adaptation through Entity-Grounded Visual Evidence for Cross-Domain Video Temporal Grounding","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2410.23728","citing_title":"GigaCheck: Detecting LLM-generated Content via Object-Centric Span Localization","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2501.14194","citing_title":"ENTER: Event Based Interpretable Reasoning for VideoQA","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17555","citing_title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03398","citing_title":"MASRA: MLLM-Assisted Semantic-Relational Consistent Alignment for Video Temporal Grounding","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27591","citing_title":"ClipTBP: Clip-Pair based Temporal Boundary Prediction with Boundary-Aware Learning for Moment Retrieval","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08966","citing_title":"How Should Video LLMs Output Time? An Analysis of Efficient Temporal Grounding Paradigms","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08522","citing_title":"UniversalVTG: A Universal and Lightweight Foundation Model for Video Temporal Grounding","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13023","citing_title":"SpotSound: Enhancing Large Audio-Language Models with Fine-Grained Temporal Grounding","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02623","citing_title":"Retrieving Any Relevant Moments: Benchmark and Models for Generalized Moment Retrieval","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3","json":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3.json","graph_json":"https://pith.science/api/pith-number/4M4MK2A6AHBHRSHU4ZLNJMHVC3/graph.json","events_json":"https://pith.science/api/pith-number/4M4MK2A6AHBHRSHU4ZLNJMHVC3/events.json","paper":"https://pith.science/paper/4M4MK2A6"},"agent_actions":{"view_html":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3","download_json":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3.json","view_paper":"https://pith.science/paper/4M4MK2A6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.08835&json=true","fetch_graph":"https://pith.science/api/pith-number/4M4MK2A6AHBHRSHU4ZLNJMHVC3/graph.json","fetch_events":"https://pith.science/api/pith-number/4M4MK2A6AHBHRSHU4ZLNJMHVC3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3/action/storage_attestation","attest_author":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3/action/author_attestation","sign_citation":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3/action/citation_signature","submit_replication":"https://pith.science/pith/4M4MK2A6AHBHRSHU4ZLNJMHVC3/action/replication_record"}},"created_at":"2026-07-05T08:39:51.453880+00:00","updated_at":"2026-07-05T08:39:51.453880+00:00"}