{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ZPWN5YNQXPU7B2QDIEL74MHFW2","short_pith_number":"pith:ZPWN5YNQ","schema_version":"1.0","canonical_sha256":"cbecdee1b0bbe9f0ea034117fe30e5b6967306eb303f049051485556a9b93593","source":{"kind":"arxiv","id":"2104.03015","version":3},"attestation_state":"computed","paper":{"title":"RTIC: Residual Learning for Text and Image Composition using Graph Convolutional Network","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Byungsoo Ko, Geonmo Gu, Minchul Shin, Yoonjae Cho","submitted_at":"2021-04-07T09:41:52Z","abstract_excerpt":"In this paper, we study the compositional learning of images and texts for image retrieval. The query is given in the form of an image and text that describes the desired modifications to the image; the goal is to retrieve the target image that satisfies the given modifications and resembles the query by composing information in both the text and image modalities. To remedy this, we propose a novel architecture designed for the image-text composition task and show that the proposed structure can effectively encode the differences between the source and target images conditioned on the text. Fu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.03015","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-04-07T09:41:52Z","cross_cats_sorted":[],"title_canon_sha256":"3e92728d1925bf601ae42bfbe45d4920babf65db7092e6618130fa8751289735","abstract_canon_sha256":"88ec312c39526c11bfc2fec85d6649f6f6eefe2146ceb229a5b802334efb1b37"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:25:30.898930Z","signature_b64":"/ZFymklo+2TZ1V38prO5BcesH5/F4FNnqx6FjXRX23hio65cTAVkqnca7Zzau0dPmrojv1c/VZkbvX031HZACA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cbecdee1b0bbe9f0ea034117fe30e5b6967306eb303f049051485556a9b93593","last_reissued_at":"2026-07-05T03:25:30.898536Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:25:30.898536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RTIC: Residual Learning for Text and Image Composition using Graph Convolutional Network","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Byungsoo Ko, Geonmo Gu, Minchul Shin, Yoonjae Cho","submitted_at":"2021-04-07T09:41:52Z","abstract_excerpt":"In this paper, we study the compositional learning of images and texts for image retrieval. The query is given in the form of an image and text that describes the desired modifications to the image; the goal is to retrieve the target image that satisfies the given modifications and resembles the query by composing information in both the text and image modalities. To remedy this, we propose a novel architecture designed for the image-text composition task and show that the proposed structure can effectively encode the differences between the source and target images conditioned on the text. Fu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.03015","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.03015/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.03015","created_at":"2026-07-05T03:25:30.898593+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.03015v3","created_at":"2026-07-05T03:25:30.898593+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.03015","created_at":"2026-07-05T03:25:30.898593+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZPWN5YNQXPU7","created_at":"2026-07-05T03:25:30.898593+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZPWN5YNQXPU7B2QD","created_at":"2026-07-05T03:25:30.898593+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZPWN5YNQ","created_at":"2026-07-05T03:25:30.898593+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.14039","citing_title":"Beyond Simple Edits: Composed Video Retrieval with Dense Modifications","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2","json":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2.json","graph_json":"https://pith.science/api/pith-number/ZPWN5YNQXPU7B2QDIEL74MHFW2/graph.json","events_json":"https://pith.science/api/pith-number/ZPWN5YNQXPU7B2QDIEL74MHFW2/events.json","paper":"https://pith.science/paper/ZPWN5YNQ"},"agent_actions":{"view_html":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2","download_json":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2.json","view_paper":"https://pith.science/paper/ZPWN5YNQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.03015&json=true","fetch_graph":"https://pith.science/api/pith-number/ZPWN5YNQXPU7B2QDIEL74MHFW2/graph.json","fetch_events":"https://pith.science/api/pith-number/ZPWN5YNQXPU7B2QDIEL74MHFW2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2/action/storage_attestation","attest_author":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2/action/author_attestation","sign_citation":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2/action/citation_signature","submit_replication":"https://pith.science/pith/ZPWN5YNQXPU7B2QDIEL74MHFW2/action/replication_record"}},"created_at":"2026-07-05T03:25:30.898593+00:00","updated_at":"2026-07-05T03:25:30.898593+00:00"}