{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:I75U6ST4DUF5WLGXKWNO5AVVNV","short_pith_number":"pith:I75U6ST4","schema_version":"1.0","canonical_sha256":"47fb4f4a7c1d0bdb2cd7559aee82b56d5f6536fb3969cc39adf44effa4d5bfc9","source":{"kind":"arxiv","id":"2212.05404","version":2},"attestation_state":"computed","paper":{"title":"Cap2Aug: Caption guided Image to Image data Augmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aniket Roy, Anirban Roy, Anshul Shah, Ketul Shah, Rama Chellappa","submitted_at":"2022-12-11T04:37:43Z","abstract_excerpt":"Visual recognition in a low-data regime is challenging and often prone to overfitting. To mitigate this issue, several data augmentation strategies have been proposed. However, standard transformations, e.g., rotation, cropping, and flipping provide limited semantic variations. To this end, we propose Cap2Aug, an image-to-image diffusion model-based data augmentation strategy using image captions as text prompts. We generate captions from the limited training images and using these captions edit the training images using an image-to-image stable diffusion model to generate semantically meaning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.05404","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-12-11T04:37:43Z","cross_cats_sorted":[],"title_canon_sha256":"2bf3f6e82e0b6eedddf3b7b35ed981142a05d8aa5f341b5b5d577f8ad3ce9afa","abstract_canon_sha256":"080b119ad5566f12d71846963a43256c21baea8677b5a5ee175a0c7d79981fe1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:44.948257Z","signature_b64":"SXMc8nUg1cawsrSdLYCVvYFZSMcll5mY+4/aPIKiNZ59m8LSatfVzmKR4cQEzNJg5Ss6okhZYOvEqbkkuWVWDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47fb4f4a7c1d0bdb2cd7559aee82b56d5f6536fb3969cc39adf44effa4d5bfc9","last_reissued_at":"2026-07-05T07:09:44.947698Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:44.947698Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cap2Aug: Caption guided Image to Image data Augmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aniket Roy, Anirban Roy, Anshul Shah, Ketul Shah, Rama Chellappa","submitted_at":"2022-12-11T04:37:43Z","abstract_excerpt":"Visual recognition in a low-data regime is challenging and often prone to overfitting. To mitigate this issue, several data augmentation strategies have been proposed. However, standard transformations, e.g., rotation, cropping, and flipping provide limited semantic variations. To this end, we propose Cap2Aug, an image-to-image diffusion model-based data augmentation strategy using image captions as text prompts. We generate captions from the limited training images and using these captions edit the training images using an image-to-image stable diffusion model to generate semantically meaning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.05404","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.05404/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.05404","created_at":"2026-07-05T07:09:44.947776+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.05404v2","created_at":"2026-07-05T07:09:44.947776+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.05404","created_at":"2026-07-05T07:09:44.947776+00:00"},{"alias_kind":"pith_short_12","alias_value":"I75U6ST4DUF5","created_at":"2026-07-05T07:09:44.947776+00:00"},{"alias_kind":"pith_short_16","alias_value":"I75U6ST4DUF5WLGX","created_at":"2026-07-05T07:09:44.947776+00:00"},{"alias_kind":"pith_short_8","alias_value":"I75U6ST4","created_at":"2026-07-05T07:09:44.947776+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV","json":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV.json","graph_json":"https://pith.science/api/pith-number/I75U6ST4DUF5WLGXKWNO5AVVNV/graph.json","events_json":"https://pith.science/api/pith-number/I75U6ST4DUF5WLGXKWNO5AVVNV/events.json","paper":"https://pith.science/paper/I75U6ST4"},"agent_actions":{"view_html":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV","download_json":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV.json","view_paper":"https://pith.science/paper/I75U6ST4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.05404&json=true","fetch_graph":"https://pith.science/api/pith-number/I75U6ST4DUF5WLGXKWNO5AVVNV/graph.json","fetch_events":"https://pith.science/api/pith-number/I75U6ST4DUF5WLGXKWNO5AVVNV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV/action/storage_attestation","attest_author":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV/action/author_attestation","sign_citation":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV/action/citation_signature","submit_replication":"https://pith.science/pith/I75U6ST4DUF5WLGXKWNO5AVVNV/action/replication_record"}},"created_at":"2026-07-05T07:09:44.947776+00:00","updated_at":"2026-07-05T07:09:44.947776+00:00"}