{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:THIKF3LMZN72ZWKH75KHOJZZL4","short_pith_number":"pith:THIKF3LM","schema_version":"1.0","canonical_sha256":"99d0a2ed6ccb7facd947ff547727395f18f16f969dc169319c26c4c8a2ecadf4","source":{"kind":"arxiv","id":"1904.11574","version":2},"attestation_state":"computed","paper":{"title":"TVQA+: Spatio-Temporal Grounding for Video Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jie Lei, Licheng Yu, Mohit Bansal, Tamara L. Berg","submitted_at":"2019-04-25T20:37:26Z","abstract_excerpt":"We present the task of Spatio-Temporal Video Question Answering, which requires intelligent systems to simultaneously retrieve relevant moments and detect referenced visual concepts (people and objects) to answer natural language questions about videos. We first augment the TVQA dataset with 310.8K bounding boxes, linking depicted objects to visual concepts in questions and answers. We name this augmented version as TVQA+. We then propose Spatio-Temporal Answerer with Grounded Evidence (STAGE), a unified framework that grounds evidence in both spatial and temporal domains to answer questions a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.11574","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-04-25T20:37:26Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"c3d7cbcc6cb7b9ee418384e301a29f8769e5621faa3256154adceaa23006d9bc","abstract_canon_sha256":"79b2e7d90f0734d969d0130d69a0d812f7de24e474e09968702ac85f506b1057"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:02:01.955132Z","signature_b64":"kh7qEfQF6QoPbFAJdGU1o+GLMSFUq3IzxaIdWCMbaZnVOVPBJyWD50ObagwNWxxCR7ifmHLF44jY8rpHu6I8Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"99d0a2ed6ccb7facd947ff547727395f18f16f969dc169319c26c4c8a2ecadf4","last_reissued_at":"2026-07-05T01:02:01.954724Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:02:01.954724Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TVQA+: Spatio-Temporal Grounding for Video Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jie Lei, Licheng Yu, Mohit Bansal, Tamara L. Berg","submitted_at":"2019-04-25T20:37:26Z","abstract_excerpt":"We present the task of Spatio-Temporal Video Question Answering, which requires intelligent systems to simultaneously retrieve relevant moments and detect referenced visual concepts (people and objects) to answer natural language questions about videos. We first augment the TVQA dataset with 310.8K bounding boxes, linking depicted objects to visual concepts in questions and answers. We name this augmented version as TVQA+. We then propose Spatio-Temporal Answerer with Grounded Evidence (STAGE), a unified framework that grounds evidence in both spatial and temporal domains to answer questions a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.11574","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.11574/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.11574","created_at":"2026-07-05T01:02:01.954777+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.11574v2","created_at":"2026-07-05T01:02:01.954777+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.11574","created_at":"2026-07-05T01:02:01.954777+00:00"},{"alias_kind":"pith_short_12","alias_value":"THIKF3LMZN72","created_at":"2026-07-05T01:02:01.954777+00:00"},{"alias_kind":"pith_short_16","alias_value":"THIKF3LMZN72ZWKH","created_at":"2026-07-05T01:02:01.954777+00:00"},{"alias_kind":"pith_short_8","alias_value":"THIKF3LM","created_at":"2026-07-05T01:02:01.954777+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13016","citing_title":"SVAG-Bench: A Large-Scale Benchmark for Multi-Instance Spatio-temporal Video Action Grounding","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4","json":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4.json","graph_json":"https://pith.science/api/pith-number/THIKF3LMZN72ZWKH75KHOJZZL4/graph.json","events_json":"https://pith.science/api/pith-number/THIKF3LMZN72ZWKH75KHOJZZL4/events.json","paper":"https://pith.science/paper/THIKF3LM"},"agent_actions":{"view_html":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4","download_json":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4.json","view_paper":"https://pith.science/paper/THIKF3LM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.11574&json=true","fetch_graph":"https://pith.science/api/pith-number/THIKF3LMZN72ZWKH75KHOJZZL4/graph.json","fetch_events":"https://pith.science/api/pith-number/THIKF3LMZN72ZWKH75KHOJZZL4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4/action/storage_attestation","attest_author":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4/action/author_attestation","sign_citation":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4/action/citation_signature","submit_replication":"https://pith.science/pith/THIKF3LMZN72ZWKH75KHOJZZL4/action/replication_record"}},"created_at":"2026-07-05T01:02:01.954777+00:00","updated_at":"2026-07-05T01:02:01.954777+00:00"}