{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UXLD5G6P2VS3IDQIPNNU7V7N5H","short_pith_number":"pith:UXLD5G6P","schema_version":"1.0","canonical_sha256":"a5d63e9bcfd565b40e087b5b4fd7ede9dd60e229add2237a6018e356d3b89034","source":{"kind":"arxiv","id":"2504.15918","version":2},"attestation_state":"computed","paper":{"title":"Ask2Loc: Learning to Locate Instructional Visual Answers by Asking Questions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CV","authors_text":"Bin Li, Chang Zong, Jian Wan, Lei Zhang, Shoujun Zhou","submitted_at":"2025-04-22T14:03:16Z","abstract_excerpt":"Locating specific segments within an instructional video is an efficient way to acquire guiding knowledge. Generally, the task of obtaining video segments for both verbal explanations and visual demonstrations is known as visual answer localization (VAL). However, users often need multiple interactions to obtain answers that align with their expectations when using the system. During these interactions, humans deepen their understanding of the video content by asking themselves questions, thereby accurately identifying the location. Therefore, we propose a new task, named In-VAL, to simulate t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.15918","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-22T14:03:16Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"61b7d927093d29a03c24355b4dae101ca1356407f6f16b3fe994ad796d333ef7","abstract_canon_sha256":"8b5e533a322d8ca2a07e730d95dd2a7e0bf014f5cb4c823bc382abbe423cb834"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:40.912769Z","signature_b64":"5qMwqaCHgg+EGM3yQNY7lWPce6uMl9gfh9Mq9+B9yXfSz1cHGZvwv5SZMdb3umjkWp/EP8FoQAKgHgfmj/1/BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a5d63e9bcfd565b40e087b5b4fd7ede9dd60e229add2237a6018e356d3b89034","last_reissued_at":"2026-07-05T10:52:40.912267Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:40.912267Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ask2Loc: Learning to Locate Instructional Visual Answers by Asking Questions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CV","authors_text":"Bin Li, Chang Zong, Jian Wan, Lei Zhang, Shoujun Zhou","submitted_at":"2025-04-22T14:03:16Z","abstract_excerpt":"Locating specific segments within an instructional video is an efficient way to acquire guiding knowledge. Generally, the task of obtaining video segments for both verbal explanations and visual demonstrations is known as visual answer localization (VAL). However, users often need multiple interactions to obtain answers that align with their expectations when using the system. During these interactions, humans deepen their understanding of the video content by asking themselves questions, thereby accurately identifying the location. Therefore, we propose a new task, named In-VAL, to simulate t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.15918","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.15918/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.15918","created_at":"2026-07-05T10:52:40.912336+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.15918v2","created_at":"2026-07-05T10:52:40.912336+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.15918","created_at":"2026-07-05T10:52:40.912336+00:00"},{"alias_kind":"pith_short_12","alias_value":"UXLD5G6P2VS3","created_at":"2026-07-05T10:52:40.912336+00:00"},{"alias_kind":"pith_short_16","alias_value":"UXLD5G6P2VS3IDQI","created_at":"2026-07-05T10:52:40.912336+00:00"},{"alias_kind":"pith_short_8","alias_value":"UXLD5G6P","created_at":"2026-07-05T10:52:40.912336+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04289","citing_title":"M$^3$-Med: A Benchmark for Multi-lingual, Multi-modal, and Multi-hop Reasoning in Medical Instructional Video Understanding","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H","json":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H.json","graph_json":"https://pith.science/api/pith-number/UXLD5G6P2VS3IDQIPNNU7V7N5H/graph.json","events_json":"https://pith.science/api/pith-number/UXLD5G6P2VS3IDQIPNNU7V7N5H/events.json","paper":"https://pith.science/paper/UXLD5G6P"},"agent_actions":{"view_html":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H","download_json":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H.json","view_paper":"https://pith.science/paper/UXLD5G6P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.15918&json=true","fetch_graph":"https://pith.science/api/pith-number/UXLD5G6P2VS3IDQIPNNU7V7N5H/graph.json","fetch_events":"https://pith.science/api/pith-number/UXLD5G6P2VS3IDQIPNNU7V7N5H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H/action/storage_attestation","attest_author":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H/action/author_attestation","sign_citation":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H/action/citation_signature","submit_replication":"https://pith.science/pith/UXLD5G6P2VS3IDQIPNNU7V7N5H/action/replication_record"}},"created_at":"2026-07-05T10:52:40.912336+00:00","updated_at":"2026-07-05T10:52:40.912336+00:00"}