{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SYB6H6M7NW6UQXIIV53ZYVSSCJ","short_pith_number":"pith:SYB6H6M7","schema_version":"1.0","canonical_sha256":"9603e3f99f6dbd485d08af779c56521251feade59fb35a73993def372071cbd6","source":{"kind":"arxiv","id":"2412.18404","version":1},"attestation_state":"computed","paper":{"title":"Extract Free Dense Misalignment from CLIP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Jeongyeon Nam, Jinbae Im, Taeho Kil, Wonjae Kim","submitted_at":"2024-12-24T12:51:05Z","abstract_excerpt":"Recent vision-language foundation models still frequently produce outputs misaligned with their inputs, evidenced by object hallucination in captioning and prompt misalignment in the text-to-image generation model. Recent studies have explored methods for identifying misaligned elements, aiming not only to enhance interpretability but also to improve model performance. However, current approaches primarily rely on large foundation models in a zero-shot manner or fine-tuned models with human annotations, which limits scalability due to significant computational costs. This work proposes a novel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18404","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-24T12:51:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"af634df6cad85c5080a60c57d952dbdb20a4d8377558dce50a8f07019655366d","abstract_canon_sha256":"e8aaae3c5df2fbe191f1e10f6c31b9db1c09626b77237002b940b8dd87e58345"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:54.896159Z","signature_b64":"6i7h5HGc2eHnRdAkQotaMjG2Op9Y8f7mF9++eaMSzW3hTIoVv1elTqlTalINasCAFxRRb/OY5fnHD//S/8LwAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9603e3f99f6dbd485d08af779c56521251feade59fb35a73993def372071cbd6","last_reissued_at":"2026-07-05T09:53:54.895726Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:54.895726Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Extract Free Dense Misalignment from CLIP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Jeongyeon Nam, Jinbae Im, Taeho Kil, Wonjae Kim","submitted_at":"2024-12-24T12:51:05Z","abstract_excerpt":"Recent vision-language foundation models still frequently produce outputs misaligned with their inputs, evidenced by object hallucination in captioning and prompt misalignment in the text-to-image generation model. Recent studies have explored methods for identifying misaligned elements, aiming not only to enhance interpretability but also to improve model performance. However, current approaches primarily rely on large foundation models in a zero-shot manner or fine-tuned models with human annotations, which limits scalability due to significant computational costs. This work proposes a novel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18404","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18404/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18404","created_at":"2026-07-05T09:53:54.895791+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18404v1","created_at":"2026-07-05T09:53:54.895791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18404","created_at":"2026-07-05T09:53:54.895791+00:00"},{"alias_kind":"pith_short_12","alias_value":"SYB6H6M7NW6U","created_at":"2026-07-05T09:53:54.895791+00:00"},{"alias_kind":"pith_short_16","alias_value":"SYB6H6M7NW6UQXII","created_at":"2026-07-05T09:53:54.895791+00:00"},{"alias_kind":"pith_short_8","alias_value":"SYB6H6M7","created_at":"2026-07-05T09:53:54.895791+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ","json":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ.json","graph_json":"https://pith.science/api/pith-number/SYB6H6M7NW6UQXIIV53ZYVSSCJ/graph.json","events_json":"https://pith.science/api/pith-number/SYB6H6M7NW6UQXIIV53ZYVSSCJ/events.json","paper":"https://pith.science/paper/SYB6H6M7"},"agent_actions":{"view_html":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ","download_json":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ.json","view_paper":"https://pith.science/paper/SYB6H6M7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18404&json=true","fetch_graph":"https://pith.science/api/pith-number/SYB6H6M7NW6UQXIIV53ZYVSSCJ/graph.json","fetch_events":"https://pith.science/api/pith-number/SYB6H6M7NW6UQXIIV53ZYVSSCJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ/action/storage_attestation","attest_author":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ/action/author_attestation","sign_citation":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ/action/citation_signature","submit_replication":"https://pith.science/pith/SYB6H6M7NW6UQXIIV53ZYVSSCJ/action/replication_record"}},"created_at":"2026-07-05T09:53:54.895791+00:00","updated_at":"2026-07-05T09:53:54.895791+00:00"}