{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:VRAZ7CMZQRDRYBC3CFYLN6ZXLX","short_pith_number":"pith:VRAZ7CMZ","schema_version":"1.0","canonical_sha256":"ac419f899984471c045b1170b6fb375df15384d840ca0835777276acc7a1cc4a","source":{"kind":"arxiv","id":"1708.02071","version":1},"attestation_state":"computed","paper":{"title":"Structured Attentions for Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Zhu, Kewei Tu, Shuaiyi Huang, Yanpeng Zhao, Yi Ma","submitted_at":"2017-08-07T11:14:11Z","abstract_excerpt":"Visual attention, which assigns weights to image regions according to their relevance to a question, is considered as an indispensable part by most Visual Question Answering models. Although the questions may involve complex relations among multiple regions, few attention models can effectively encode such cross-region relations. In this paper, we demonstrate the importance of encoding such relations by showing the limited effective receptive field of ResNet on two datasets, and propose to model the visual attention as a multivariate distribution over a grid-structured Conditional Random Field"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1708.02071","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2017-08-07T11:14:11Z","cross_cats_sorted":[],"title_canon_sha256":"df8a991811fe18d07348a7bc7cb03a37561a623529c2f87abc08eb313fe0bacb","abstract_canon_sha256":"10cde01da9f62744e1d4d17913237dc24cbbf8224ccae5d3c4de6f5593490655"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:38:32.393937Z","signature_b64":"/Pxq7kkj8vI4A0UijHNTrFOMLfKYZ8CxP2uJt6TwgXrf2cyayodu2xIPyaS5h0k09wGS+DCcPaU47PXBPSVIDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac419f899984471c045b1170b6fb375df15384d840ca0835777276acc7a1cc4a","last_reissued_at":"2026-05-18T00:38:32.393418Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:38:32.393418Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Structured Attentions for Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Zhu, Kewei Tu, Shuaiyi Huang, Yanpeng Zhao, Yi Ma","submitted_at":"2017-08-07T11:14:11Z","abstract_excerpt":"Visual attention, which assigns weights to image regions according to their relevance to a question, is considered as an indispensable part by most Visual Question Answering models. Although the questions may involve complex relations among multiple regions, few attention models can effectively encode such cross-region relations. In this paper, we demonstrate the importance of encoding such relations by showing the limited effective receptive field of ResNet on two datasets, and propose to model the visual attention as a multivariate distribution over a grid-structured Conditional Random Field"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1708.02071","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1708.02071","created_at":"2026-05-18T00:38:32.393513+00:00"},{"alias_kind":"arxiv_version","alias_value":"1708.02071v1","created_at":"2026-05-18T00:38:32.393513+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1708.02071","created_at":"2026-05-18T00:38:32.393513+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRAZ7CMZQRDR","created_at":"2026-05-18T12:31:49.984773+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRAZ7CMZQRDRYBC3","created_at":"2026-05-18T12:31:49.984773+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRAZ7CMZ","created_at":"2026-05-18T12:31:49.984773+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX","json":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX.json","graph_json":"https://pith.science/api/pith-number/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/graph.json","events_json":"https://pith.science/api/pith-number/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/events.json","paper":"https://pith.science/paper/VRAZ7CMZ"},"agent_actions":{"view_html":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX","download_json":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX.json","view_paper":"https://pith.science/paper/VRAZ7CMZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1708.02071&json=true","fetch_graph":"https://pith.science/api/pith-number/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/graph.json","fetch_events":"https://pith.science/api/pith-number/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/action/storage_attestation","attest_author":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/action/author_attestation","sign_citation":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/action/citation_signature","submit_replication":"https://pith.science/pith/VRAZ7CMZQRDRYBC3CFYLN6ZXLX/action/replication_record"}},"created_at":"2026-05-18T00:38:32.393513+00:00","updated_at":"2026-05-18T00:38:32.393513+00:00"}