{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZCGUROLFN34QG3SG5HUTPTFQOZ","short_pith_number":"pith:ZCGUROLF","schema_version":"1.0","canonical_sha256":"c88d48b9656ef9036e46e9e937ccb0764dae54f740a5c4ed23193111d2bd0fe5","source":{"kind":"arxiv","id":"2206.00629","version":2},"attestation_state":"computed","paper":{"title":"CLIP4IDC: CLIP for Image Difference Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jorma Laaksonen, Tzu-Jui Julius Wang, Zixin Guo","submitted_at":"2022-06-01T17:02:08Z","abstract_excerpt":"Image Difference Captioning (IDC) aims at generating sentences to describe differences between two similar-looking images. Conventional approaches learn an IDC model with a pre-trained and usually frozen visual feature extractor. Accordingly, two major issues may arise: (1) a large domain gap usually exists between the pre-training datasets used for training such a visual encoder and that of the downstream IDC task, and (2) the visual feature extractor, when separately encoding two images, often does not effectively encode the visual changes between two images. Due to the excellent zero-shot p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.00629","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-06-01T17:02:08Z","cross_cats_sorted":[],"title_canon_sha256":"7883bad010ad8bb17cd0234e1c914250d9c781de3183905eb9ecfccfb97f8061","abstract_canon_sha256":"75f179e9c7dc85841a71c5a31cf1da69eb35fa057a8c245a5921afbe65865b70"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:07:45.316058Z","signature_b64":"P4OEbSNDc2BcRZXULqDfgfSRxD1iXkq7v36hMXkPKtW02XM1Dren6gN6Je6TQosCt798uRz9a5eGjnDw7gfiAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c88d48b9656ef9036e46e9e937ccb0764dae54f740a5c4ed23193111d2bd0fe5","last_reissued_at":"2026-07-05T05:07:45.315594Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:07:45.315594Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIP4IDC: CLIP for Image Difference Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jorma Laaksonen, Tzu-Jui Julius Wang, Zixin Guo","submitted_at":"2022-06-01T17:02:08Z","abstract_excerpt":"Image Difference Captioning (IDC) aims at generating sentences to describe differences between two similar-looking images. Conventional approaches learn an IDC model with a pre-trained and usually frozen visual feature extractor. Accordingly, two major issues may arise: (1) a large domain gap usually exists between the pre-training datasets used for training such a visual encoder and that of the downstream IDC task, and (2) the visual feature extractor, when separately encoding two images, often does not effectively encode the visual changes between two images. Due to the excellent zero-shot p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.00629","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.00629/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.00629","created_at":"2026-07-05T05:07:45.315649+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.00629v2","created_at":"2026-07-05T05:07:45.315649+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.00629","created_at":"2026-07-05T05:07:45.315649+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZCGUROLFN34Q","created_at":"2026-07-05T05:07:45.315649+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZCGUROLFN34QG3SG","created_at":"2026-07-05T05:07:45.315649+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZCGUROLF","created_at":"2026-07-05T05:07:45.315649+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25445","citing_title":"C3-Bench: A Context-Aware Change Captioning Benchmark","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ","json":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ.json","graph_json":"https://pith.science/api/pith-number/ZCGUROLFN34QG3SG5HUTPTFQOZ/graph.json","events_json":"https://pith.science/api/pith-number/ZCGUROLFN34QG3SG5HUTPTFQOZ/events.json","paper":"https://pith.science/paper/ZCGUROLF"},"agent_actions":{"view_html":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ","download_json":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ.json","view_paper":"https://pith.science/paper/ZCGUROLF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.00629&json=true","fetch_graph":"https://pith.science/api/pith-number/ZCGUROLFN34QG3SG5HUTPTFQOZ/graph.json","fetch_events":"https://pith.science/api/pith-number/ZCGUROLFN34QG3SG5HUTPTFQOZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ/action/storage_attestation","attest_author":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ/action/author_attestation","sign_citation":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ/action/citation_signature","submit_replication":"https://pith.science/pith/ZCGUROLFN34QG3SG5HUTPTFQOZ/action/replication_record"}},"created_at":"2026-07-05T05:07:45.315649+00:00","updated_at":"2026-07-05T05:07:45.315649+00:00"}