{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:36JYG46PPUSVSESFG4LAZLQ4II","short_pith_number":"pith:36JYG46P","schema_version":"1.0","canonical_sha256":"df938373cf7d2559124537160cae1c4209408aac125702db136dbf5d53c50fb0","source":{"kind":"arxiv","id":"1908.05054","version":2},"attestation_state":"computed","paper":{"title":"Fusion of Detected Objects in Text for Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chris Alberti, David Reitter, Jeffrey Ling, Michael Collins","submitted_at":"2019-08-14T10:03:12Z","abstract_excerpt":"To advance models of multimodal context, we introduce a simple yet powerful neural architecture for data that combines vision and natural language. The \"Bounding Boxes in Text Transformer\" (B2T2) also leverages referential information binding words to portions of the image in a single unified architecture. B2T2 is highly effective on the Visual Commonsense Reasoning benchmark (https://visualcommonsense.com), achieving a new state-of-the-art with a 25% relative reduction in error rate compared to published baselines and obtaining the best performance to date on the public leaderboard (as of May"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.05054","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-08-14T10:03:12Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"12f5630026bd217c46e1eb1cae6cd3c1b584a5356a46ae87a4b764d40a604074","abstract_canon_sha256":"a8966c950167690f01897724030ba49599779f3c860281a1ae7ff4db5a2c857e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:16:36.159849Z","signature_b64":"+kmQhrJ3LYsmnTi8tpdNhdbBLN8H66GdaTS/zFkwwP5A+WMymIbkP7N3p/VL49r0SNHDdat1C9GjgXXohsIHDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df938373cf7d2559124537160cae1c4209408aac125702db136dbf5d53c50fb0","last_reissued_at":"2026-07-05T00:16:36.159371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:16:36.159371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fusion of Detected Objects in Text for Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chris Alberti, David Reitter, Jeffrey Ling, Michael Collins","submitted_at":"2019-08-14T10:03:12Z","abstract_excerpt":"To advance models of multimodal context, we introduce a simple yet powerful neural architecture for data that combines vision and natural language. The \"Bounding Boxes in Text Transformer\" (B2T2) also leverages referential information binding words to portions of the image in a single unified architecture. B2T2 is highly effective on the Visual Commonsense Reasoning benchmark (https://visualcommonsense.com), achieving a new state-of-the-art with a 25% relative reduction in error rate compared to published baselines and obtaining the best performance to date on the public leaderboard (as of May"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.05054","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.05054/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.05054","created_at":"2026-07-05T00:16:36.159432+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.05054v2","created_at":"2026-07-05T00:16:36.159432+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.05054","created_at":"2026-07-05T00:16:36.159432+00:00"},{"alias_kind":"pith_short_12","alias_value":"36JYG46PPUSV","created_at":"2026-07-05T00:16:36.159432+00:00"},{"alias_kind":"pith_short_16","alias_value":"36JYG46PPUSVSESF","created_at":"2026-07-05T00:16:36.159432+00:00"},{"alias_kind":"pith_short_8","alias_value":"36JYG46P","created_at":"2026-07-05T00:16:36.159432+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2308.08089","citing_title":"DragNUWA: Fine-grained Control in Video Generation by Integrating Text, Image, and Trajectory","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II","json":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II.json","graph_json":"https://pith.science/api/pith-number/36JYG46PPUSVSESFG4LAZLQ4II/graph.json","events_json":"https://pith.science/api/pith-number/36JYG46PPUSVSESFG4LAZLQ4II/events.json","paper":"https://pith.science/paper/36JYG46P"},"agent_actions":{"view_html":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II","download_json":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II.json","view_paper":"https://pith.science/paper/36JYG46P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.05054&json=true","fetch_graph":"https://pith.science/api/pith-number/36JYG46PPUSVSESFG4LAZLQ4II/graph.json","fetch_events":"https://pith.science/api/pith-number/36JYG46PPUSVSESFG4LAZLQ4II/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II/action/timestamp_anchor","attest_storage":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II/action/storage_attestation","attest_author":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II/action/author_attestation","sign_citation":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II/action/citation_signature","submit_replication":"https://pith.science/pith/36JYG46PPUSVSESFG4LAZLQ4II/action/replication_record"}},"created_at":"2026-07-05T00:16:36.159432+00:00","updated_at":"2026-07-05T00:16:36.159432+00:00"}