{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:2QUFDLOARBZOB5BPIPJOOMEYMC","short_pith_number":"pith:2QUFDLOA","schema_version":"1.0","canonical_sha256":"d42851adc08872e0f42f43d2e7309860a1c8b62dfe6dd9aac32a940e0a34d232","source":{"kind":"arxiv","id":"2010.03182","version":3},"attestation_state":"computed","paper":{"title":"VICTR: Visual Information Captured Text Representation for Text-to-Image Multimodal Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Josiah Poon, Kunze Wang, Siqu Long, Siwen Luo, Soyeon Caren Han","submitted_at":"2020-10-07T05:25:30Z","abstract_excerpt":"Text-to-image multimodal tasks, generating/retrieving an image from a given text description, are extremely challenging tasks since raw text descriptions cover quite limited information in order to fully describe visually realistic images. We propose a new visual contextual text representation for text-to-image multimodal tasks, VICTR, which captures rich visual semantic information of objects from the text input. First, we use the text description as initial input and conduct dependency parsing to extract the syntactic structure and analyse the semantic aspect, including object quantities, to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.03182","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-10-07T05:25:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"970963db177db828dcb0808087464fe10cfa2fdb38130c396ddcc6052dd2d44c","abstract_canon_sha256":"cc8c632e3710739cf6ce3c9a8f1810c172ad57e2b1f9d1a13b494c63effef479"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:45:45.367987Z","signature_b64":"TbAQCBaPKEMB9WUz3/675cxQIJgxuWo1ITISFJi3Sy2/aARccCHpcWJGxqSqe2LnohtWk9OCOt+OMA3dBvgWBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d42851adc08872e0f42f43d2e7309860a1c8b62dfe6dd9aac32a940e0a34d232","last_reissued_at":"2026-07-05T01:45:45.367580Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:45:45.367580Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VICTR: Visual Information Captured Text Representation for Text-to-Image Multimodal Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Josiah Poon, Kunze Wang, Siqu Long, Siwen Luo, Soyeon Caren Han","submitted_at":"2020-10-07T05:25:30Z","abstract_excerpt":"Text-to-image multimodal tasks, generating/retrieving an image from a given text description, are extremely challenging tasks since raw text descriptions cover quite limited information in order to fully describe visually realistic images. We propose a new visual contextual text representation for text-to-image multimodal tasks, VICTR, which captures rich visual semantic information of objects from the text input. First, we use the text description as initial input and conduct dependency parsing to extract the syntactic structure and analyse the semantic aspect, including object quantities, to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.03182","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.03182/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.03182","created_at":"2026-07-05T01:45:45.367642+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.03182v3","created_at":"2026-07-05T01:45:45.367642+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.03182","created_at":"2026-07-05T01:45:45.367642+00:00"},{"alias_kind":"pith_short_12","alias_value":"2QUFDLOARBZO","created_at":"2026-07-05T01:45:45.367642+00:00"},{"alias_kind":"pith_short_16","alias_value":"2QUFDLOARBZOB5BP","created_at":"2026-07-05T01:45:45.367642+00:00"},{"alias_kind":"pith_short_8","alias_value":"2QUFDLOA","created_at":"2026-07-05T01:45:45.367642+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.14045","citing_title":"From Image Captioning to Visual Storytelling","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC","json":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC.json","graph_json":"https://pith.science/api/pith-number/2QUFDLOARBZOB5BPIPJOOMEYMC/graph.json","events_json":"https://pith.science/api/pith-number/2QUFDLOARBZOB5BPIPJOOMEYMC/events.json","paper":"https://pith.science/paper/2QUFDLOA"},"agent_actions":{"view_html":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC","download_json":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC.json","view_paper":"https://pith.science/paper/2QUFDLOA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.03182&json=true","fetch_graph":"https://pith.science/api/pith-number/2QUFDLOARBZOB5BPIPJOOMEYMC/graph.json","fetch_events":"https://pith.science/api/pith-number/2QUFDLOARBZOB5BPIPJOOMEYMC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC/action/storage_attestation","attest_author":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC/action/author_attestation","sign_citation":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC/action/citation_signature","submit_replication":"https://pith.science/pith/2QUFDLOARBZOB5BPIPJOOMEYMC/action/replication_record"}},"created_at":"2026-07-05T01:45:45.367642+00:00","updated_at":"2026-07-05T01:45:45.367642+00:00"}