{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:FNGCTSYUKOADZR5VEOM2C25HVZ","short_pith_number":"pith:FNGCTSYU","schema_version":"1.0","canonical_sha256":"2b4c29cb1453803cc7b52399a16ba7ae76de15c60c85af6f78e9c9a8758b44cc","source":{"kind":"arxiv","id":"2203.08013","version":2},"attestation_state":"computed","paper":{"title":"End-to-End Modeling via Information Tree for One-Shot Natural Language Spatial Video Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fei Wu, Haoyu Zhang, Jiaxu Miao, Jin Wang, Mengze Li, Peng Wang, Shengyu Zhang, Shiliang Pu, Tianbao Wang, Wenming Tan, Wenqiao Zhang, Zhou Zhao","submitted_at":"2022-03-15T15:50:45Z","abstract_excerpt":"Natural language spatial video grounding aims to detect the relevant objects in video frames with descriptive sentences as the query. In spite of the great advances, most existing methods rely on dense video frame annotations, which require a tremendous amount of human effort. To achieve effective grounding under a limited annotation budget, we investigate one-shot video grounding, and learn to ground natural language in all video frames with solely one frame labeled, in an end-to-end manner. One major challenge of end-to-end one-shot video grounding is the existence of videos frames that are "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.08013","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-15T15:50:45Z","cross_cats_sorted":[],"title_canon_sha256":"3646259b7efcedcb355c8bcdc48c4217179ab57f98f6cadf72049a6585de2826","abstract_canon_sha256":"7ac48d7480a942daaa2d13378d4795c243e1a67bcc93da851f8094882726f638"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:25:16.428343Z","signature_b64":"RLY971rfLpvO8NluxWgZ3GiU4eU3Ae/ZrfthkjWL591DwMZZ5Cu50k8UrSEnqws937neqkPLMwoSf2N3Mgn9Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b4c29cb1453803cc7b52399a16ba7ae76de15c60c85af6f78e9c9a8758b44cc","last_reissued_at":"2026-07-05T04:25:16.427928Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:25:16.427928Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-End Modeling via Information Tree for One-Shot Natural Language Spatial Video Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fei Wu, Haoyu Zhang, Jiaxu Miao, Jin Wang, Mengze Li, Peng Wang, Shengyu Zhang, Shiliang Pu, Tianbao Wang, Wenming Tan, Wenqiao Zhang, Zhou Zhao","submitted_at":"2022-03-15T15:50:45Z","abstract_excerpt":"Natural language spatial video grounding aims to detect the relevant objects in video frames with descriptive sentences as the query. In spite of the great advances, most existing methods rely on dense video frame annotations, which require a tremendous amount of human effort. To achieve effective grounding under a limited annotation budget, we investigate one-shot video grounding, and learn to ground natural language in all video frames with solely one frame labeled, in an end-to-end manner. One major challenge of end-to-end one-shot video grounding is the existence of videos frames that are "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.08013","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.08013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.08013","created_at":"2026-07-05T04:25:16.427984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.08013v2","created_at":"2026-07-05T04:25:16.427984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.08013","created_at":"2026-07-05T04:25:16.427984+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNGCTSYUKOAD","created_at":"2026-07-05T04:25:16.427984+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNGCTSYUKOADZR5V","created_at":"2026-07-05T04:25:16.427984+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNGCTSYU","created_at":"2026-07-05T04:25:16.427984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.23940","citing_title":"Graft: Integrating the Domain Knowledge via Efficient Parameter Synergy for MLLMs","ref_index":2019,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ","json":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ.json","graph_json":"https://pith.science/api/pith-number/FNGCTSYUKOADZR5VEOM2C25HVZ/graph.json","events_json":"https://pith.science/api/pith-number/FNGCTSYUKOADZR5VEOM2C25HVZ/events.json","paper":"https://pith.science/paper/FNGCTSYU"},"agent_actions":{"view_html":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ","download_json":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ.json","view_paper":"https://pith.science/paper/FNGCTSYU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.08013&json=true","fetch_graph":"https://pith.science/api/pith-number/FNGCTSYUKOADZR5VEOM2C25HVZ/graph.json","fetch_events":"https://pith.science/api/pith-number/FNGCTSYUKOADZR5VEOM2C25HVZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ/action/storage_attestation","attest_author":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ/action/author_attestation","sign_citation":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ/action/citation_signature","submit_replication":"https://pith.science/pith/FNGCTSYUKOADZR5VEOM2C25HVZ/action/replication_record"}},"created_at":"2026-07-05T04:25:16.427984+00:00","updated_at":"2026-07-05T04:25:16.427984+00:00"}