{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FH5SOBGWACRKQLTYY3VUFNQFNC","short_pith_number":"pith:FH5SOBGW","schema_version":"1.0","canonical_sha256":"29fb2704d600a2a82e78c6eb42b60568990b45341665d1a680e0b27f09a13892","source":{"kind":"arxiv","id":"2511.01390","version":2},"attestation_state":"computed","paper":{"title":"SEPS: Semantic-enhanced Patch Slimming Framework for fine-grained cross-modal alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Haoji Zhang, Junsi Li, Ming Sun, Xinyu Mao, Yu Liang","submitted_at":"2025-11-03T09:41:32Z","abstract_excerpt":"Fine-grained cross-modal alignment aims to establish precise local correspondences between vision and language, forming a cornerstone for visual question answering and related multimodal applications. Current approaches face challenges in addressing patch redundancy and ambiguity, which arise from the inherent information density disparities across modalities. Recently, Multimodal Large Language Models (MLLMs) have emerged as promising solutions to bridge this gap through their robust semantic generation capabilities. However, the dense textual outputs from MLLMs may introduce conflicts with t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2511.01390","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-11-03T09:41:32Z","cross_cats_sorted":["cs.AI","cs.MM"],"title_canon_sha256":"202a8c85152d77db85b1691b7f29d2c3e5fe11bf29b87ecc3ad1e789ef8d04d6","abstract_canon_sha256":"6ca8646230b38c67ca5c951355c2f622eb14cfda5d5407c72468d71439e1f192"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-03T01:17:14.338671Z","signature_b64":"DLiVYWJNOygPh/KH+hLqlmV383JinfX/q1l7IpkkvRYHCzeZ7qJklDNqcNzq6CU3Kf0p3Ss5oh0oBHCUPAhTBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29fb2704d600a2a82e78c6eb42b60568990b45341665d1a680e0b27f09a13892","last_reissued_at":"2026-07-03T01:17:14.338221Z","signature_status":"signed_v1","first_computed_at":"2026-07-03T01:17:14.338221Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SEPS: Semantic-enhanced Patch Slimming Framework for fine-grained cross-modal alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.MM"],"primary_cat":"cs.CV","authors_text":"Haoji Zhang, Junsi Li, Ming Sun, Xinyu Mao, Yu Liang","submitted_at":"2025-11-03T09:41:32Z","abstract_excerpt":"Fine-grained cross-modal alignment aims to establish precise local correspondences between vision and language, forming a cornerstone for visual question answering and related multimodal applications. Current approaches face challenges in addressing patch redundancy and ambiguity, which arise from the inherent information density disparities across modalities. Recently, Multimodal Large Language Models (MLLMs) have emerged as promising solutions to bridge this gap through their robust semantic generation capabilities. However, the dense textual outputs from MLLMs may introduce conflicts with t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2511.01390","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2511.01390/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2511.01390","created_at":"2026-07-03T01:17:14.338278+00:00"},{"alias_kind":"arxiv_version","alias_value":"2511.01390v2","created_at":"2026-07-03T01:17:14.338278+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2511.01390","created_at":"2026-07-03T01:17:14.338278+00:00"},{"alias_kind":"pith_short_12","alias_value":"FH5SOBGWACRK","created_at":"2026-07-03T01:17:14.338278+00:00"},{"alias_kind":"pith_short_16","alias_value":"FH5SOBGWACRKQLTY","created_at":"2026-07-03T01:17:14.338278+00:00"},{"alias_kind":"pith_short_8","alias_value":"FH5SOBGW","created_at":"2026-07-03T01:17:14.338278+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC","json":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC.json","graph_json":"https://pith.science/api/pith-number/FH5SOBGWACRKQLTYY3VUFNQFNC/graph.json","events_json":"https://pith.science/api/pith-number/FH5SOBGWACRKQLTYY3VUFNQFNC/events.json","paper":"https://pith.science/paper/FH5SOBGW"},"agent_actions":{"view_html":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC","download_json":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC.json","view_paper":"https://pith.science/paper/FH5SOBGW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2511.01390&json=true","fetch_graph":"https://pith.science/api/pith-number/FH5SOBGWACRKQLTYY3VUFNQFNC/graph.json","fetch_events":"https://pith.science/api/pith-number/FH5SOBGWACRKQLTYY3VUFNQFNC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC/action/storage_attestation","attest_author":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC/action/author_attestation","sign_citation":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC/action/citation_signature","submit_replication":"https://pith.science/pith/FH5SOBGWACRKQLTYY3VUFNQFNC/action/replication_record"}},"created_at":"2026-07-03T01:17:14.338278+00:00","updated_at":"2026-07-03T01:17:14.338278+00:00"}