{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ZOMVHGSXWJVG3SW6IUHNR4YE5I","short_pith_number":"pith:ZOMVHGSX","schema_version":"1.0","canonical_sha256":"cb99539a57b26a6dcade450ed8f304ea1128e5b5e6899f0b043259fcf68a8616","source":{"kind":"arxiv","id":"2109.10678","version":1},"attestation_state":"computed","paper":{"title":"Natural Language Video Localization with Learnable Moment Proposals","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jian Shao, Jun Xiao, Long Chen, Shaoning Xiao, Yueting Zhuang","submitted_at":"2021-09-22T12:18:58Z","abstract_excerpt":"Given an untrimmed video and a natural language query, Natural Language Video Localization (NLVL) aims to identify the video moment described by the query. To address this task, existing methods can be roughly grouped into two groups: 1) propose-and-rank models first define a set of hand-designed moment candidates and then find out the best-matching one. 2) proposal-free models directly predict two temporal boundaries of the referential moment from frames. Currently, almost all the propose-and-rank methods have inferior performance than proposal-free counterparts. In this paper, we argue that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.10678","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-09-22T12:18:58Z","cross_cats_sorted":[],"title_canon_sha256":"a4c70315b747c8005d0984b32980db883e1eefa665829466c9400301c2909243","abstract_canon_sha256":"2b0d6dd9c60ba9c8cd753a60aa1e60288a5f1b3e5436fcb7f8d565e396153155"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:12:08.966759Z","signature_b64":"WizOR67x6NHx/9zCXCsgPvDKS2swJb/YS15m0rOnmY/l//QVng8F73ZXeDBMJ/g43SBRzNl4VEXwb+skp2/EAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb99539a57b26a6dcade450ed8f304ea1128e5b5e6899f0b043259fcf68a8616","last_reissued_at":"2026-07-05T05:12:08.966347Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:12:08.966347Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Natural Language Video Localization with Learnable Moment Proposals","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jian Shao, Jun Xiao, Long Chen, Shaoning Xiao, Yueting Zhuang","submitted_at":"2021-09-22T12:18:58Z","abstract_excerpt":"Given an untrimmed video and a natural language query, Natural Language Video Localization (NLVL) aims to identify the video moment described by the query. To address this task, existing methods can be roughly grouped into two groups: 1) propose-and-rank models first define a set of hand-designed moment candidates and then find out the best-matching one. 2) proposal-free models directly predict two temporal boundaries of the referential moment from frames. Currently, almost all the propose-and-rank methods have inferior performance than proposal-free counterparts. In this paper, we argue that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.10678","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.10678/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.10678","created_at":"2026-07-05T05:12:08.966405+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.10678v1","created_at":"2026-07-05T05:12:08.966405+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.10678","created_at":"2026-07-05T05:12:08.966405+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZOMVHGSXWJVG","created_at":"2026-07-05T05:12:08.966405+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZOMVHGSXWJVG3SW6","created_at":"2026-07-05T05:12:08.966405+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZOMVHGSX","created_at":"2026-07-05T05:12:08.966405+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2312.02549","citing_title":"DemaFormer: Damped Exponential Moving Average Transformer with Energy-Based Modeling for Temporal Language Grounding","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2412.07157","citing_title":"Multi-Scale Contrastive Learning for Video Temporal Grounding","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I","json":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I.json","graph_json":"https://pith.science/api/pith-number/ZOMVHGSXWJVG3SW6IUHNR4YE5I/graph.json","events_json":"https://pith.science/api/pith-number/ZOMVHGSXWJVG3SW6IUHNR4YE5I/events.json","paper":"https://pith.science/paper/ZOMVHGSX"},"agent_actions":{"view_html":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I","download_json":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I.json","view_paper":"https://pith.science/paper/ZOMVHGSX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.10678&json=true","fetch_graph":"https://pith.science/api/pith-number/ZOMVHGSXWJVG3SW6IUHNR4YE5I/graph.json","fetch_events":"https://pith.science/api/pith-number/ZOMVHGSXWJVG3SW6IUHNR4YE5I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I/action/storage_attestation","attest_author":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I/action/author_attestation","sign_citation":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I/action/citation_signature","submit_replication":"https://pith.science/pith/ZOMVHGSXWJVG3SW6IUHNR4YE5I/action/replication_record"}},"created_at":"2026-07-05T05:12:08.966405+00:00","updated_at":"2026-07-05T05:12:08.966405+00:00"}