{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:FXI6DBKG7WCEK5JWJTSQ5K7F4C","short_pith_number":"pith:FXI6DBKG","schema_version":"1.0","canonical_sha256":"2dd1e18546fd844575364ce50eabe5e0b6b2353283b2e308c8cdd4726c174fb0","source":{"kind":"arxiv","id":"2204.10938","version":3},"attestation_state":"computed","paper":{"title":"A Multi-level Alignment Training Scheme for Video-and-Language Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Feiyang Niu, Govind Thattai, Qing Ping, Yubo Zhang","submitted_at":"2022-04-22T21:46:52Z","abstract_excerpt":"To solve video-and-language grounding tasks, the key is for the network to understand the connection between the two modalities. For a pair of video and language description, their semantic relation is reflected by their encodings' similarity. A good multi-modality encoder should be able to well capture both inputs' semantics and encode them in the shared feature space where embedding distance gets properly translated into their semantic similarity. In this work, we focused on this semantic connection between video and language, and developed a multi-level alignment training scheme to directly"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.10938","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-22T21:46:52Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"abe4493d689cb48786699531778cafd317fdecf104714db3bfa1edb62f3100c1","abstract_canon_sha256":"0512729a8621e0d2a56a100ffd75f694d2a570a2b356e1fc0e6e8f05820b0d92"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:45:27.773774Z","signature_b64":"T17KuBbn25WsgqhczY4PXdlzMn/+pG+B7zuTWcusQT9mF8kuwKBY2eekQoRS9VxVtqBrV+MhZr2sP3XJvbK9Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dd1e18546fd844575364ce50eabe5e0b6b2353283b2e308c8cdd4726c174fb0","last_reissued_at":"2026-07-05T05:45:27.773406Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:45:27.773406Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Multi-level Alignment Training Scheme for Video-and-Language Grounding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Feiyang Niu, Govind Thattai, Qing Ping, Yubo Zhang","submitted_at":"2022-04-22T21:46:52Z","abstract_excerpt":"To solve video-and-language grounding tasks, the key is for the network to understand the connection between the two modalities. For a pair of video and language description, their semantic relation is reflected by their encodings' similarity. A good multi-modality encoder should be able to well capture both inputs' semantics and encode them in the shared feature space where embedding distance gets properly translated into their semantic similarity. In this work, we focused on this semantic connection between video and language, and developed a multi-level alignment training scheme to directly"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.10938","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.10938/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.10938","created_at":"2026-07-05T05:45:27.773462+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.10938v3","created_at":"2026-07-05T05:45:27.773462+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.10938","created_at":"2026-07-05T05:45:27.773462+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXI6DBKG7WCE","created_at":"2026-07-05T05:45:27.773462+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXI6DBKG7WCEK5JW","created_at":"2026-07-05T05:45:27.773462+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXI6DBKG","created_at":"2026-07-05T05:45:27.773462+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C","json":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C.json","graph_json":"https://pith.science/api/pith-number/FXI6DBKG7WCEK5JWJTSQ5K7F4C/graph.json","events_json":"https://pith.science/api/pith-number/FXI6DBKG7WCEK5JWJTSQ5K7F4C/events.json","paper":"https://pith.science/paper/FXI6DBKG"},"agent_actions":{"view_html":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C","download_json":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C.json","view_paper":"https://pith.science/paper/FXI6DBKG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.10938&json=true","fetch_graph":"https://pith.science/api/pith-number/FXI6DBKG7WCEK5JWJTSQ5K7F4C/graph.json","fetch_events":"https://pith.science/api/pith-number/FXI6DBKG7WCEK5JWJTSQ5K7F4C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C/action/storage_attestation","attest_author":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C/action/author_attestation","sign_citation":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C/action/citation_signature","submit_replication":"https://pith.science/pith/FXI6DBKG7WCEK5JWJTSQ5K7F4C/action/replication_record"}},"created_at":"2026-07-05T05:45:27.773462+00:00","updated_at":"2026-07-05T05:45:27.773462+00:00"}