{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6GGJHM6IH2YORJXYO3V7CFBLF5","short_pith_number":"pith:6GGJHM6I","schema_version":"1.0","canonical_sha256":"f18c93b3c83eb0e8a6f876ebf1142b2f5034e9c00705d3f652c3979f837c2607","source":{"kind":"arxiv","id":"2105.07122","version":3},"attestation_state":"computed","paper":{"title":"Premise-based Multimodal Reasoning: Conditional Inference on Joint Textual and Visual Clues","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haoran Meng, Heming Xia, Lin Xu, Qingxiu Dong, Shoujie Tong, Sujian Li, Tian Feng, Tianyu Liu, Weidong Zhan, Zhongyu Wei, Ziwei Qin, Zuifang Sui","submitted_at":"2021-05-15T03:25:42Z","abstract_excerpt":"It is a common practice for recent works in vision language cross-modal reasoning to adopt a binary or multi-choice classification formulation taking as input a set of source image(s) and textual query. In this work, we take a sober look at such an unconditional formulation in the sense that no prior knowledge is specified with respect to the source image(s). Inspired by the designs of both visual commonsense reasoning and natural language inference tasks, we propose a new task termed Premise-based Multi-modal Reasoning(PMR) where a textual premise is the background presumption on each source "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.07122","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-05-15T03:25:42Z","cross_cats_sorted":[],"title_canon_sha256":"68cf4dc3e214a257a38c44404af339a82adddedcdc7d2f0c8d90910989588fd7","abstract_canon_sha256":"78f9e8af9edc43385d88ca14ad988774d02fd2cbe6d01bc342ce7f20ac3cf795"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:58.532152Z","signature_b64":"DCFgug1FEZ+gJ9qbdAP4zY7FvD1/j1y91BJI0dOzmqQ7IerHzDoieQXwH04vA3ut3ktmF0RqUK9rA0m2XRTPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f18c93b3c83eb0e8a6f876ebf1142b2f5034e9c00705d3f652c3979f837c2607","last_reissued_at":"2026-07-05T04:05:58.531734Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:58.531734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Premise-based Multimodal Reasoning: Conditional Inference on Joint Textual and Visual Clues","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haoran Meng, Heming Xia, Lin Xu, Qingxiu Dong, Shoujie Tong, Sujian Li, Tian Feng, Tianyu Liu, Weidong Zhan, Zhongyu Wei, Ziwei Qin, Zuifang Sui","submitted_at":"2021-05-15T03:25:42Z","abstract_excerpt":"It is a common practice for recent works in vision language cross-modal reasoning to adopt a binary or multi-choice classification formulation taking as input a set of source image(s) and textual query. In this work, we take a sober look at such an unconditional formulation in the sense that no prior knowledge is specified with respect to the source image(s). Inspired by the designs of both visual commonsense reasoning and natural language inference tasks, we propose a new task termed Premise-based Multi-modal Reasoning(PMR) where a textual premise is the background presumption on each source "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.07122","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.07122/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.07122","created_at":"2026-07-05T04:05:58.531791+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.07122v3","created_at":"2026-07-05T04:05:58.531791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.07122","created_at":"2026-07-05T04:05:58.531791+00:00"},{"alias_kind":"pith_short_12","alias_value":"6GGJHM6IH2YO","created_at":"2026-07-05T04:05:58.531791+00:00"},{"alias_kind":"pith_short_16","alias_value":"6GGJHM6IH2YORJXY","created_at":"2026-07-05T04:05:58.531791+00:00"},{"alias_kind":"pith_short_8","alias_value":"6GGJHM6I","created_at":"2026-07-05T04:05:58.531791+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5","json":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5.json","graph_json":"https://pith.science/api/pith-number/6GGJHM6IH2YORJXYO3V7CFBLF5/graph.json","events_json":"https://pith.science/api/pith-number/6GGJHM6IH2YORJXYO3V7CFBLF5/events.json","paper":"https://pith.science/paper/6GGJHM6I"},"agent_actions":{"view_html":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5","download_json":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5.json","view_paper":"https://pith.science/paper/6GGJHM6I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.07122&json=true","fetch_graph":"https://pith.science/api/pith-number/6GGJHM6IH2YORJXYO3V7CFBLF5/graph.json","fetch_events":"https://pith.science/api/pith-number/6GGJHM6IH2YORJXYO3V7CFBLF5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5/action/storage_attestation","attest_author":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5/action/author_attestation","sign_citation":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5/action/citation_signature","submit_replication":"https://pith.science/pith/6GGJHM6IH2YORJXYO3V7CFBLF5/action/replication_record"}},"created_at":"2026-07-05T04:05:58.531791+00:00","updated_at":"2026-07-05T04:05:58.531791+00:00"}