{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:AQMYMVOSQIGJM2IRF7X6U6YWEQ","short_pith_number":"pith:AQMYMVOS","schema_version":"1.0","canonical_sha256":"04198655d2820c9669112fefea7b16243e16cf17e165dab115f67a91a5e53fb6","source":{"kind":"arxiv","id":"2009.11278","version":1},"attestation_state":"computed","paper":{"title":"X-LXMERT: Paint, Caption and Answer Questions with Multi-Modal Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aniruddha Kembhavi, Dustin Schwenk, Hannaneh Hajishirzi, Jaemin Cho, Jiasen Lu","submitted_at":"2020-09-23T17:45:17Z","abstract_excerpt":"Mirroring the success of masked language models, vision-and-language counterparts like ViLBERT, LXMERT and UNITER have achieved state of the art performance on a variety of multimodal discriminative tasks like visual question answering and visual grounding. Recent work has also successfully adapted such models towards the generative task of image captioning. This begs the question: Can these models go the other way and generate images from pieces of text? Our analysis of a popular representative from this model family - LXMERT - finds that it is unable to generate rich and semantically meaning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2009.11278","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-09-23T17:45:17Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"ccbe493651f78be9fb049758a8592f59869f33afb4fb703a7d82a449b940f66f","abstract_canon_sha256":"2c69c1f22915fb391bd164f120760c7b611f200ceca47593f0c3355dd5d2c2df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:37:38.086218Z","signature_b64":"m8O5B7QxOx1C9CwQkVjHZOP2SLxUVRmWr/WHhHLQdJVP5VEFVjEDbDouzALn2Eh0K9YGCCrX6gTrsdYWE1zOCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04198655d2820c9669112fefea7b16243e16cf17e165dab115f67a91a5e53fb6","last_reissued_at":"2026-07-05T01:37:38.085777Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:37:38.085777Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"X-LXMERT: Paint, Caption and Answer Questions with Multi-Modal Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aniruddha Kembhavi, Dustin Schwenk, Hannaneh Hajishirzi, Jaemin Cho, Jiasen Lu","submitted_at":"2020-09-23T17:45:17Z","abstract_excerpt":"Mirroring the success of masked language models, vision-and-language counterparts like ViLBERT, LXMERT and UNITER have achieved state of the art performance on a variety of multimodal discriminative tasks like visual question answering and visual grounding. Recent work has also successfully adapted such models towards the generative task of image captioning. This begs the question: Can these models go the other way and generate images from pieces of text? Our analysis of a popular representative from this model family - LXMERT - finds that it is unable to generate rich and semantically meaning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2009.11278","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2009.11278/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2009.11278","created_at":"2026-07-05T01:37:38.085837+00:00"},{"alias_kind":"arxiv_version","alias_value":"2009.11278v1","created_at":"2026-07-05T01:37:38.085837+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2009.11278","created_at":"2026-07-05T01:37:38.085837+00:00"},{"alias_kind":"pith_short_12","alias_value":"AQMYMVOSQIGJ","created_at":"2026-07-05T01:37:38.085837+00:00"},{"alias_kind":"pith_short_16","alias_value":"AQMYMVOSQIGJM2IR","created_at":"2026-07-05T01:37:38.085837+00:00"},{"alias_kind":"pith_short_8","alias_value":"AQMYMVOS","created_at":"2026-07-05T01:37:38.085837+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2206.10789","citing_title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ","json":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ.json","graph_json":"https://pith.science/api/pith-number/AQMYMVOSQIGJM2IRF7X6U6YWEQ/graph.json","events_json":"https://pith.science/api/pith-number/AQMYMVOSQIGJM2IRF7X6U6YWEQ/events.json","paper":"https://pith.science/paper/AQMYMVOS"},"agent_actions":{"view_html":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ","download_json":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ.json","view_paper":"https://pith.science/paper/AQMYMVOS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2009.11278&json=true","fetch_graph":"https://pith.science/api/pith-number/AQMYMVOSQIGJM2IRF7X6U6YWEQ/graph.json","fetch_events":"https://pith.science/api/pith-number/AQMYMVOSQIGJM2IRF7X6U6YWEQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ/action/storage_attestation","attest_author":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ/action/author_attestation","sign_citation":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ/action/citation_signature","submit_replication":"https://pith.science/pith/AQMYMVOSQIGJM2IRF7X6U6YWEQ/action/replication_record"}},"created_at":"2026-07-05T01:37:38.085837+00:00","updated_at":"2026-07-05T01:37:38.085837+00:00"}