{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GAFRA6VFYD2D7Y7C4Y3B4Y7SAV","short_pith_number":"pith:GAFRA6VF","schema_version":"1.0","canonical_sha256":"300b107aa5c0f43fe3e2e6361e63f2054921ea3f02f25d82ab6da7ace2700372","source":{"kind":"arxiv","id":"2302.08063","version":2},"attestation_state":"computed","paper":{"title":"MINOTAUR: Multi-task Video Grounding From Multimodal Queries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Du Tran, Effrosyni Mavroudi, Leonid Sigal, Lorenzo Torresani, Matt Feiszli, Raghav Goyal, Sainbayar Sukhbaatar, Xitong Yang","submitted_at":"2023-02-16T04:00:03Z","abstract_excerpt":"Video understanding tasks take many forms, from action detection to visual query localization and spatio-temporal grounding of sentences. These tasks differ in the type of inputs (only video, or video-query pair where query is an image region or sentence) and outputs (temporal segments or spatio-temporal tubes). However, at their core they require the same fundamental understanding of the video, i.e., the actors and objects in it, their actions and interactions. So far these tasks have been tackled in isolation with individual, highly specialized architectures, which do not exploit the interpl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08063","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-02-16T04:00:03Z","cross_cats_sorted":[],"title_canon_sha256":"cac8b165d3890f7e07a12328bdceb8de3b093bb1af727f691523805940f41f0f","abstract_canon_sha256":"a1b2b53779efd199c1d22685bb017ed586a493f40026ea875e99bf7e30cae116"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:52:26.216787Z","signature_b64":"5I3/IYfojoqZklKeDbZXJjJy6T4E5SNImxbePGrqDzo07HaEfwdZfm/WVngGtISBltw5H6ZMQte0DRsUvHuJAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"300b107aa5c0f43fe3e2e6361e63f2054921ea3f02f25d82ab6da7ace2700372","last_reissued_at":"2026-07-05T05:52:26.216283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:52:26.216283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MINOTAUR: Multi-task Video Grounding From Multimodal Queries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Du Tran, Effrosyni Mavroudi, Leonid Sigal, Lorenzo Torresani, Matt Feiszli, Raghav Goyal, Sainbayar Sukhbaatar, Xitong Yang","submitted_at":"2023-02-16T04:00:03Z","abstract_excerpt":"Video understanding tasks take many forms, from action detection to visual query localization and spatio-temporal grounding of sentences. These tasks differ in the type of inputs (only video, or video-query pair where query is an image region or sentence) and outputs (temporal segments or spatio-temporal tubes). However, at their core they require the same fundamental understanding of the video, i.e., the actors and objects in it, their actions and interactions. So far these tasks have been tackled in isolation with individual, highly specialized architectures, which do not exploit the interpl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08063","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08063/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08063","created_at":"2026-07-05T05:52:26.216343+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08063v2","created_at":"2026-07-05T05:52:26.216343+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08063","created_at":"2026-07-05T05:52:26.216343+00:00"},{"alias_kind":"pith_short_12","alias_value":"GAFRA6VFYD2D","created_at":"2026-07-05T05:52:26.216343+00:00"},{"alias_kind":"pith_short_16","alias_value":"GAFRA6VFYD2D7Y7C","created_at":"2026-07-05T05:52:26.216343+00:00"},{"alias_kind":"pith_short_8","alias_value":"GAFRA6VF","created_at":"2026-07-05T05:52:26.216343+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.11901","citing_title":"BSN-II: The First Light Curve Study of Eight Total Eclipsing Contact Binary Stars with Shallow Fillout Factors","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV","json":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV.json","graph_json":"https://pith.science/api/pith-number/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/graph.json","events_json":"https://pith.science/api/pith-number/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/events.json","paper":"https://pith.science/paper/GAFRA6VF"},"agent_actions":{"view_html":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV","download_json":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV.json","view_paper":"https://pith.science/paper/GAFRA6VF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08063&json=true","fetch_graph":"https://pith.science/api/pith-number/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/graph.json","fetch_events":"https://pith.science/api/pith-number/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/action/storage_attestation","attest_author":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/action/author_attestation","sign_citation":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/action/citation_signature","submit_replication":"https://pith.science/pith/GAFRA6VFYD2D7Y7C4Y3B4Y7SAV/action/replication_record"}},"created_at":"2026-07-05T05:52:26.216343+00:00","updated_at":"2026-07-05T05:52:26.216343+00:00"}