{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KA4RD346R3HMMFPX3IXPN22NYI","short_pith_number":"pith:KA4RD346","schema_version":"1.0","canonical_sha256":"503911ef9e8ecec615f7da2ef6eb4dc201365b89b8b5fe03d02c50591c233926","source":{"kind":"arxiv","id":"2312.08548","version":1},"attestation_state":"computed","paper":{"title":"EVP: Enhanced Visual Perception using Inverse Multi-Attentive Feature Refinement and Regularized Image-Text Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Matthias M\\\"uller, Mykola Lavreniuk, Peter Wonka, Shariq Farooq Bhat","submitted_at":"2023-12-13T22:20:45Z","abstract_excerpt":"This work presents the network architecture EVP (Enhanced Visual Perception). EVP builds on the previous work VPD which paved the way to use the Stable Diffusion network for computer vision tasks. We propose two major enhancements. First, we develop the Inverse Multi-Attentive Feature Refinement (IMAFR) module which enhances feature learning capabilities by aggregating spatial information from higher pyramid levels. Second, we propose a novel image-text alignment module for improved feature extraction of the Stable Diffusion backbone. The resulting architecture is suitable for a wide variety o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.08548","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-13T22:20:45Z","cross_cats_sorted":[],"title_canon_sha256":"5fa75716e69277258968780e7e3b24022c0cf23b5eab596d7d2d4457cda432c5","abstract_canon_sha256":"20e2dc073cd907e70bbcfff9827f785cc1558606b2f4dcefcbc7e046728e613f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:24:05.850674Z","signature_b64":"ppbHyJ+DT4pv8Nuzlw6nEIto+V+SHsuV31FlVI7qHw1H0QV/S0tKdHSzXomhFbyPwiM50rBUD6tq5YTFHprkBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"503911ef9e8ecec615f7da2ef6eb4dc201365b89b8b5fe03d02c50591c233926","last_reissued_at":"2026-07-05T07:24:05.850213Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:24:05.850213Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EVP: Enhanced Visual Perception using Inverse Multi-Attentive Feature Refinement and Regularized Image-Text Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Matthias M\\\"uller, Mykola Lavreniuk, Peter Wonka, Shariq Farooq Bhat","submitted_at":"2023-12-13T22:20:45Z","abstract_excerpt":"This work presents the network architecture EVP (Enhanced Visual Perception). EVP builds on the previous work VPD which paved the way to use the Stable Diffusion network for computer vision tasks. We propose two major enhancements. First, we develop the Inverse Multi-Attentive Feature Refinement (IMAFR) module which enhances feature learning capabilities by aggregating spatial information from higher pyramid levels. Second, we propose a novel image-text alignment module for improved feature extraction of the Stable Diffusion backbone. The resulting architecture is suitable for a wide variety o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.08548","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.08548/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.08548","created_at":"2026-07-05T07:24:05.850275+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.08548v1","created_at":"2026-07-05T07:24:05.850275+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.08548","created_at":"2026-07-05T07:24:05.850275+00:00"},{"alias_kind":"pith_short_12","alias_value":"KA4RD346R3HM","created_at":"2026-07-05T07:24:05.850275+00:00"},{"alias_kind":"pith_short_16","alias_value":"KA4RD346R3HMMFPX","created_at":"2026-07-05T07:24:05.850275+00:00"},{"alias_kind":"pith_short_8","alias_value":"KA4RD346","created_at":"2026-07-05T07:24:05.850275+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.03339","citing_title":"Hierarchical Awareness Adapters with Hybrid Pyramid Feature Fusion for Dense Depth Prediction","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07664","citing_title":"Monocular Depth Estimation From the Perspective of Feature Restoration: A Diffusion Enhanced Depth Restoration Approach","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI","json":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI.json","graph_json":"https://pith.science/api/pith-number/KA4RD346R3HMMFPX3IXPN22NYI/graph.json","events_json":"https://pith.science/api/pith-number/KA4RD346R3HMMFPX3IXPN22NYI/events.json","paper":"https://pith.science/paper/KA4RD346"},"agent_actions":{"view_html":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI","download_json":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI.json","view_paper":"https://pith.science/paper/KA4RD346","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.08548&json=true","fetch_graph":"https://pith.science/api/pith-number/KA4RD346R3HMMFPX3IXPN22NYI/graph.json","fetch_events":"https://pith.science/api/pith-number/KA4RD346R3HMMFPX3IXPN22NYI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI/action/storage_attestation","attest_author":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI/action/author_attestation","sign_citation":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI/action/citation_signature","submit_replication":"https://pith.science/pith/KA4RD346R3HMMFPX3IXPN22NYI/action/replication_record"}},"created_at":"2026-07-05T07:24:05.850275+00:00","updated_at":"2026-07-05T07:24:05.850275+00:00"}