{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:DARFH2INSL2QRM5IDHE6UDA6HA","short_pith_number":"pith:DARFH2IN","schema_version":"1.0","canonical_sha256":"182253e90d92f508b3a819c9ea0c1e3812ab572c7d673de4087acb8def0e8656","source":{"kind":"arxiv","id":"1903.03166","version":2},"attestation_state":"computed","paper":{"title":"CLEVR-Dialog: A Diagnostic Dataset for Multi-Round Reasoning in Visual Dialog","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Devi Parikh, Dhruv Batra, Jos\\'e M. F. Moura, Marcus Rohrbach, Satwik Kottur","submitted_at":"2019-03-07T20:18:39Z","abstract_excerpt":"Visual Dialog is a multimodal task of answering a sequence of questions grounded in an image, using the conversation history as context. It entails challenges in vision, language, reasoning, and grounding. However, studying these subtasks in isolation on large, real datasets is infeasible as it requires prohibitively-expensive complete annotation of the 'state' of all images and dialogs.\n  We develop CLEVR-Dialog, a large diagnostic dataset for studying multi-round reasoning in visual dialog. Specifically, we construct a dialog grammar that is grounded in the scene graphs of the images from th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1903.03166","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-03-07T20:18:39Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"e902fd8fecd2335fba4e294f83b029c6fd4c0a75fc4f70a5432ee06e12ca5a27","abstract_canon_sha256":"ee8306b552dd8e92507fd50756914f975bb8ea372df1a5933ab92d5fff5f2d18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:05:34.132023Z","signature_b64":"nWk3trX5NoEe+oHLt+ZKvFBALVGpP48bXBR+PPrmkfToctF+X1rebs8rZ8XIAaze/8AM/CEOzA/9CupIXZvQDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"182253e90d92f508b3a819c9ea0c1e3812ab572c7d673de4087acb8def0e8656","last_reissued_at":"2026-07-05T00:05:34.131541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:05:34.131541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLEVR-Dialog: A Diagnostic Dataset for Multi-Round Reasoning in Visual Dialog","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Devi Parikh, Dhruv Batra, Jos\\'e M. F. Moura, Marcus Rohrbach, Satwik Kottur","submitted_at":"2019-03-07T20:18:39Z","abstract_excerpt":"Visual Dialog is a multimodal task of answering a sequence of questions grounded in an image, using the conversation history as context. It entails challenges in vision, language, reasoning, and grounding. However, studying these subtasks in isolation on large, real datasets is infeasible as it requires prohibitively-expensive complete annotation of the 'state' of all images and dialogs.\n  We develop CLEVR-Dialog, a large diagnostic dataset for studying multi-round reasoning in visual dialog. Specifically, we construct a dialog grammar that is grounded in the scene graphs of the images from th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1903.03166","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1903.03166/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1903.03166","created_at":"2026-07-05T00:05:34.131601+00:00"},{"alias_kind":"arxiv_version","alias_value":"1903.03166v2","created_at":"2026-07-05T00:05:34.131601+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1903.03166","created_at":"2026-07-05T00:05:34.131601+00:00"},{"alias_kind":"pith_short_12","alias_value":"DARFH2INSL2Q","created_at":"2026-07-05T00:05:34.131601+00:00"},{"alias_kind":"pith_short_16","alias_value":"DARFH2INSL2QRM5I","created_at":"2026-07-05T00:05:34.131601+00:00"},{"alias_kind":"pith_short_8","alias_value":"DARFH2IN","created_at":"2026-07-05T00:05:34.131601+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24020","citing_title":"Machine Intelligence that Understands Visual and Linguistic Information and Interacts with Humans and Environments","ref_index":111,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA","json":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA.json","graph_json":"https://pith.science/api/pith-number/DARFH2INSL2QRM5IDHE6UDA6HA/graph.json","events_json":"https://pith.science/api/pith-number/DARFH2INSL2QRM5IDHE6UDA6HA/events.json","paper":"https://pith.science/paper/DARFH2IN"},"agent_actions":{"view_html":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA","download_json":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA.json","view_paper":"https://pith.science/paper/DARFH2IN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1903.03166&json=true","fetch_graph":"https://pith.science/api/pith-number/DARFH2INSL2QRM5IDHE6UDA6HA/graph.json","fetch_events":"https://pith.science/api/pith-number/DARFH2INSL2QRM5IDHE6UDA6HA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA/action/storage_attestation","attest_author":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA/action/author_attestation","sign_citation":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA/action/citation_signature","submit_replication":"https://pith.science/pith/DARFH2INSL2QRM5IDHE6UDA6HA/action/replication_record"}},"created_at":"2026-07-05T00:05:34.131601+00:00","updated_at":"2026-07-05T00:05:34.131601+00:00"}