{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IERXBEKVQSODU572BPSOA2RZM6","short_pith_number":"pith:IERXBEKV","schema_version":"1.0","canonical_sha256":"4123709155849c3a77fa0be4e06a3967a633e2b51c1304a2e0dffb037b69f489","source":{"kind":"arxiv","id":"2312.13503","version":1},"attestation_state":"computed","paper":{"title":"InfoVisDial: An Informative Visual Dialogue Dataset by Bridging Large Multimodal and Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bill Howe, Bingbing Wen, Jianfeng Wang, Lijuan Wang, Zhe Gan, Zhengyuan Yang","submitted_at":"2023-12-21T00:44:45Z","abstract_excerpt":"In this paper, we build a visual dialogue dataset, named InfoVisDial, which provides rich informative answers in each round even with external knowledge related to the visual content. Different from existing datasets where the answer is compact and short, InfoVisDial contains long free-form answers with rich information in each round of dialogue. For effective data collection, the key idea is to bridge the large-scale multimodal model (e.g., GIT) and the language models (e.g., GPT-3). GIT can describe the image content even with scene text, while GPT-3 can generate informative dialogue based o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.13503","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-21T00:44:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fd99d85ca298a04942d889fef776fa08b2593b886b64eacee38cef402c793a18","abstract_canon_sha256":"f9b8f7ac1ddac249b802f48c591a053b0e5c2d2f88d80d1dfd5c1f2044105a76"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:26:45.542069Z","signature_b64":"7cVprFgKD3B0m75oVjkgstfJkdMy10BUPhG4DpvD6YuTf3FIpJ+OsGiR3rYA5n9xdS1EC6Gf5jIq1K9GcXcdDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4123709155849c3a77fa0be4e06a3967a633e2b51c1304a2e0dffb037b69f489","last_reissued_at":"2026-07-05T07:26:45.541608Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:26:45.541608Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InfoVisDial: An Informative Visual Dialogue Dataset by Bridging Large Multimodal and Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bill Howe, Bingbing Wen, Jianfeng Wang, Lijuan Wang, Zhe Gan, Zhengyuan Yang","submitted_at":"2023-12-21T00:44:45Z","abstract_excerpt":"In this paper, we build a visual dialogue dataset, named InfoVisDial, which provides rich informative answers in each round even with external knowledge related to the visual content. Different from existing datasets where the answer is compact and short, InfoVisDial contains long free-form answers with rich information in each round of dialogue. For effective data collection, the key idea is to bridge the large-scale multimodal model (e.g., GIT) and the language models (e.g., GPT-3). GIT can describe the image content even with scene text, while GPT-3 can generate informative dialogue based o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.13503","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.13503/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.13503","created_at":"2026-07-05T07:26:45.541658+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.13503v1","created_at":"2026-07-05T07:26:45.541658+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.13503","created_at":"2026-07-05T07:26:45.541658+00:00"},{"alias_kind":"pith_short_12","alias_value":"IERXBEKVQSOD","created_at":"2026-07-05T07:26:45.541658+00:00"},{"alias_kind":"pith_short_16","alias_value":"IERXBEKVQSODU572","created_at":"2026-07-05T07:26:45.541658+00:00"},{"alias_kind":"pith_short_8","alias_value":"IERXBEKV","created_at":"2026-07-05T07:26:45.541658+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18322","citing_title":"Escaping the SpuriVerse: Can Large Vision-Language Models Generalize Beyond Seen Spurious Correlations?","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6","json":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6.json","graph_json":"https://pith.science/api/pith-number/IERXBEKVQSODU572BPSOA2RZM6/graph.json","events_json":"https://pith.science/api/pith-number/IERXBEKVQSODU572BPSOA2RZM6/events.json","paper":"https://pith.science/paper/IERXBEKV"},"agent_actions":{"view_html":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6","download_json":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6.json","view_paper":"https://pith.science/paper/IERXBEKV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.13503&json=true","fetch_graph":"https://pith.science/api/pith-number/IERXBEKVQSODU572BPSOA2RZM6/graph.json","fetch_events":"https://pith.science/api/pith-number/IERXBEKVQSODU572BPSOA2RZM6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6/action/storage_attestation","attest_author":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6/action/author_attestation","sign_citation":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6/action/citation_signature","submit_replication":"https://pith.science/pith/IERXBEKVQSODU572BPSOA2RZM6/action/replication_record"}},"created_at":"2026-07-05T07:26:45.541658+00:00","updated_at":"2026-07-05T07:26:45.541658+00:00"}