{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JRVMGDLRB52A2VH3RRZRQ2CZ36","short_pith_number":"pith:JRVMGDLR","schema_version":"1.0","canonical_sha256":"4c6ac30d710f740d54fb8c73186859dfb28bd6a73cd46337807aaaabe9d9ffe1","source":{"kind":"arxiv","id":"2407.05645","version":4},"attestation_state":"computed","paper":{"title":"OneDiff: A Generalist Model for Image Difference Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Erdong Hu, Jing Liu, Longteng Guo, Shuning Xue, Tongtian Yue, Zijia Zhao","submitted_at":"2024-07-08T06:14:37Z","abstract_excerpt":"In computer vision, Image Difference Captioning (IDC) is crucial for accurately describing variations between closely related images. Traditional IDC methods often rely on specialist models, which restrict their applicability across varied contexts. This paper introduces the OneDiff model, a novel generalist approach that utilizes a robust vision-language model architecture, integrating a siamese image encoder with a Visual Delta Module. This innovative configuration allows for the precise detection and articulation of fine-grained differences between image pairs. OneDiff is trained through a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.05645","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-08T06:14:37Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"17773f75493f63c2872bfd433248e2b0567a6ad5a569a6688eebd179ba2b6156","abstract_canon_sha256":"733e89ffebbb221b28428faebe89647629a65dab5e5da9a8261e32f0747119e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:38.336761Z","signature_b64":"eZ6zsjgAMzVL+FTF2NL4s9U7h5XL+A1Oi6oStZ6SZz7wyfsc0ZQ7eEIR0tkGRVxc9Rst1RS5rmeDEjsvvbZGAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c6ac30d710f740d54fb8c73186859dfb28bd6a73cd46337807aaaabe9d9ffe1","last_reissued_at":"2026-07-05T11:08:38.336265Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:38.336265Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OneDiff: A Generalist Model for Image Difference Captioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Erdong Hu, Jing Liu, Longteng Guo, Shuning Xue, Tongtian Yue, Zijia Zhao","submitted_at":"2024-07-08T06:14:37Z","abstract_excerpt":"In computer vision, Image Difference Captioning (IDC) is crucial for accurately describing variations between closely related images. Traditional IDC methods often rely on specialist models, which restrict their applicability across varied contexts. This paper introduces the OneDiff model, a novel generalist approach that utilizes a robust vision-language model architecture, integrating a siamese image encoder with a Visual Delta Module. This innovative configuration allows for the precise detection and articulation of fine-grained differences between image pairs. OneDiff is trained through a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.05645","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.05645/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.05645","created_at":"2026-07-05T11:08:38.336324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.05645v4","created_at":"2026-07-05T11:08:38.336324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.05645","created_at":"2026-07-05T11:08:38.336324+00:00"},{"alias_kind":"pith_short_12","alias_value":"JRVMGDLRB52A","created_at":"2026-07-05T11:08:38.336324+00:00"},{"alias_kind":"pith_short_16","alias_value":"JRVMGDLRB52A2VH3","created_at":"2026-07-05T11:08:38.336324+00:00"},{"alias_kind":"pith_short_8","alias_value":"JRVMGDLR","created_at":"2026-07-05T11:08:38.336324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25445","citing_title":"C3-Bench: A Context-Aware Change Captioning Benchmark","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36","json":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36.json","graph_json":"https://pith.science/api/pith-number/JRVMGDLRB52A2VH3RRZRQ2CZ36/graph.json","events_json":"https://pith.science/api/pith-number/JRVMGDLRB52A2VH3RRZRQ2CZ36/events.json","paper":"https://pith.science/paper/JRVMGDLR"},"agent_actions":{"view_html":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36","download_json":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36.json","view_paper":"https://pith.science/paper/JRVMGDLR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.05645&json=true","fetch_graph":"https://pith.science/api/pith-number/JRVMGDLRB52A2VH3RRZRQ2CZ36/graph.json","fetch_events":"https://pith.science/api/pith-number/JRVMGDLRB52A2VH3RRZRQ2CZ36/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36/action/storage_attestation","attest_author":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36/action/author_attestation","sign_citation":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36/action/citation_signature","submit_replication":"https://pith.science/pith/JRVMGDLRB52A2VH3RRZRQ2CZ36/action/replication_record"}},"created_at":"2026-07-05T11:08:38.336324+00:00","updated_at":"2026-07-05T11:08:38.336324+00:00"}