{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:336M7JKWJKYRL7TFFTC6ZRDGTG","short_pith_number":"pith:336M7JKW","schema_version":"1.0","canonical_sha256":"defccfa5564ab115fe652cc5ecc466998f0a5232731a8c2211b2aca5b23802bf","source":{"kind":"arxiv","id":"2410.13666","version":1},"attestation_state":"computed","paper":{"title":"VL-GLUE: A Suite of Fundamental yet Challenging Visuo-Linguistic Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chitta Baral, Kartik Aggarwal, Mandy Zhou, Mutsumi Nakamura, Shailaja Keyur Sampat, Shankar Kailas, Yezhou Yang","submitted_at":"2024-10-17T15:27:17Z","abstract_excerpt":"Deriving inference from heterogeneous inputs (such as images, text, and audio) is an important skill for humans to perform day-to-day tasks. A similar ability is desirable for the development of advanced Artificial Intelligence (AI) systems. While state-of-the-art models are rapidly closing the gap with human-level performance on diverse computer vision and NLP tasks separately, they struggle to solve tasks that require joint reasoning over visual and textual modalities. Inspired by GLUE (Wang et. al., 2018)- a multitask benchmark for natural language understanding, we propose VL-GLUE in this "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13666","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-17T15:27:17Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"3fb21ccad1f60e4a30ace0a29296cb972435d2ec82a640374cb80983dbd2e547","abstract_canon_sha256":"ff4877bf557b8ecd3deaebe267d672078717496edb677da891e82cb937002dbe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:04.153972Z","signature_b64":"FmNwa0Zhkk/2cb1LQn0zRMrO2ImYpb1YGjDRoaGO4aNT88GeEMLUyClMYqC0L5vfO/CSJeFrc233JjbMUfL6CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"defccfa5564ab115fe652cc5ecc466998f0a5232731a8c2211b2aca5b23802bf","last_reissued_at":"2026-07-05T09:22:04.153499Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:04.153499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VL-GLUE: A Suite of Fundamental yet Challenging Visuo-Linguistic Reasoning Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chitta Baral, Kartik Aggarwal, Mandy Zhou, Mutsumi Nakamura, Shailaja Keyur Sampat, Shankar Kailas, Yezhou Yang","submitted_at":"2024-10-17T15:27:17Z","abstract_excerpt":"Deriving inference from heterogeneous inputs (such as images, text, and audio) is an important skill for humans to perform day-to-day tasks. A similar ability is desirable for the development of advanced Artificial Intelligence (AI) systems. While state-of-the-art models are rapidly closing the gap with human-level performance on diverse computer vision and NLP tasks separately, they struggle to solve tasks that require joint reasoning over visual and textual modalities. Inspired by GLUE (Wang et. al., 2018)- a multitask benchmark for natural language understanding, we propose VL-GLUE in this "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13666","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13666/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13666","created_at":"2026-07-05T09:22:04.153559+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13666v1","created_at":"2026-07-05T09:22:04.153559+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13666","created_at":"2026-07-05T09:22:04.153559+00:00"},{"alias_kind":"pith_short_12","alias_value":"336M7JKWJKYR","created_at":"2026-07-05T09:22:04.153559+00:00"},{"alias_kind":"pith_short_16","alias_value":"336M7JKWJKYRL7TF","created_at":"2026-07-05T09:22:04.153559+00:00"},{"alias_kind":"pith_short_8","alias_value":"336M7JKW","created_at":"2026-07-05T09:22:04.153559+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG","json":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG.json","graph_json":"https://pith.science/api/pith-number/336M7JKWJKYRL7TFFTC6ZRDGTG/graph.json","events_json":"https://pith.science/api/pith-number/336M7JKWJKYRL7TFFTC6ZRDGTG/events.json","paper":"https://pith.science/paper/336M7JKW"},"agent_actions":{"view_html":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG","download_json":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG.json","view_paper":"https://pith.science/paper/336M7JKW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13666&json=true","fetch_graph":"https://pith.science/api/pith-number/336M7JKWJKYRL7TFFTC6ZRDGTG/graph.json","fetch_events":"https://pith.science/api/pith-number/336M7JKWJKYRL7TFFTC6ZRDGTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG/action/storage_attestation","attest_author":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG/action/author_attestation","sign_citation":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG/action/citation_signature","submit_replication":"https://pith.science/pith/336M7JKWJKYRL7TFFTC6ZRDGTG/action/replication_record"}},"created_at":"2026-07-05T09:22:04.153559+00:00","updated_at":"2026-07-05T09:22:04.153559+00:00"}