{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:4NUCC5CMGGHDSXIGLE6LKKO7XK","short_pith_number":"pith:4NUCC5CM","schema_version":"1.0","canonical_sha256":"e36821744c318e395d06593cb529dfba88158ae0acf6279320a22a387c633b3c","source":{"kind":"arxiv","id":"2109.03415","version":1},"attestation_state":"computed","paper":{"title":"Vision Matters When It Should: Sanity Checking Multimodal Machine Translation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Duygu Ataman, Jiaoda Li, Rico Sennrich","submitted_at":"2021-09-08T03:32:48Z","abstract_excerpt":"Multimodal machine translation (MMT) systems have been shown to outperform their text-only neural machine translation (NMT) counterparts when visual context is available. However, recent studies have also shown that the performance of MMT models is only marginally impacted when the associated image is replaced with an unrelated image or noise, which suggests that the visual context might not be exploited by the model at all. We hypothesize that this might be caused by the nature of the commonly used evaluation benchmark, also known as Multi30K, where the translations of image captions were pre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.03415","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-09-08T03:32:48Z","cross_cats_sorted":[],"title_canon_sha256":"44aaa8716fce2aa901bcc06f17b892ca8b60373a02130a8fc90d19c37eec577c","abstract_canon_sha256":"11e30a23b070379b95a792bba274e5879d3f7ccdf4c30930de200761cc263023"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:12:30.882447Z","signature_b64":"GmLUZO03HhRDZdlEvY/8kChEE91DDNsPD98I2Ui/2/1DL0KRQN5DrCXZLNQZVUf3D8zKQH6/LjihEiMGLKqnAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e36821744c318e395d06593cb529dfba88158ae0acf6279320a22a387c633b3c","last_reissued_at":"2026-07-05T03:12:30.882077Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:12:30.882077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision Matters When It Should: Sanity Checking Multimodal Machine Translation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Duygu Ataman, Jiaoda Li, Rico Sennrich","submitted_at":"2021-09-08T03:32:48Z","abstract_excerpt":"Multimodal machine translation (MMT) systems have been shown to outperform their text-only neural machine translation (NMT) counterparts when visual context is available. However, recent studies have also shown that the performance of MMT models is only marginally impacted when the associated image is replaced with an unrelated image or noise, which suggests that the visual context might not be exploited by the model at all. We hypothesize that this might be caused by the nature of the commonly used evaluation benchmark, also known as Multi30K, where the translations of image captions were pre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.03415","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.03415/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.03415","created_at":"2026-07-05T03:12:30.882140+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.03415v1","created_at":"2026-07-05T03:12:30.882140+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.03415","created_at":"2026-07-05T03:12:30.882140+00:00"},{"alias_kind":"pith_short_12","alias_value":"4NUCC5CMGGHD","created_at":"2026-07-05T03:12:30.882140+00:00"},{"alias_kind":"pith_short_16","alias_value":"4NUCC5CMGGHDSXIG","created_at":"2026-07-05T03:12:30.882140+00:00"},{"alias_kind":"pith_short_8","alias_value":"4NUCC5CM","created_at":"2026-07-05T03:12:30.882140+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.05714","citing_title":"TopicVD: A Topic-Based Dataset of Video-Guided Multimodal Machine Translation for Documentaries","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK","json":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK.json","graph_json":"https://pith.science/api/pith-number/4NUCC5CMGGHDSXIGLE6LKKO7XK/graph.json","events_json":"https://pith.science/api/pith-number/4NUCC5CMGGHDSXIGLE6LKKO7XK/events.json","paper":"https://pith.science/paper/4NUCC5CM"},"agent_actions":{"view_html":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK","download_json":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK.json","view_paper":"https://pith.science/paper/4NUCC5CM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.03415&json=true","fetch_graph":"https://pith.science/api/pith-number/4NUCC5CMGGHDSXIGLE6LKKO7XK/graph.json","fetch_events":"https://pith.science/api/pith-number/4NUCC5CMGGHDSXIGLE6LKKO7XK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK/action/storage_attestation","attest_author":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK/action/author_attestation","sign_citation":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK/action/citation_signature","submit_replication":"https://pith.science/pith/4NUCC5CMGGHDSXIGLE6LKKO7XK/action/replication_record"}},"created_at":"2026-07-05T03:12:30.882140+00:00","updated_at":"2026-07-05T03:12:30.882140+00:00"}