{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QDUTESFMWGQNQS7NRN5CEYZ6VG","short_pith_number":"pith:QDUTESFM","schema_version":"1.0","canonical_sha256":"80e93248acb1a0d84bed8b7a22633ea9a6116b48348c19b954f01c30b5ae42a9","source":{"kind":"arxiv","id":"2412.06774","version":1},"attestation_state":"computed","paper":{"title":"Visual Lexicon: Rich Image Features in Language Space","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alireza Fathi, Cordelia Schmid, Trevor Darrell, Xingyi Zhou, Xudong Wang","submitted_at":"2024-12-09T18:57:24Z","abstract_excerpt":"We present Visual Lexicon, a novel visual language that encodes rich image information into the text space of vocabulary tokens while retaining intricate visual details that are often challenging to convey in natural language. Unlike traditional methods that prioritize either high-level semantics (e.g., CLIP) or pixel-level reconstruction (e.g., VAE), ViLex simultaneously captures rich semantic content and fine visual details, enabling high-quality image generation and comprehensive visual scene understanding. Through a self-supervised learning pipeline, ViLex generates tokens optimized for re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.06774","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-09T18:57:24Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8eaca33bb080f5e42752e401cc803ce8def637c41ecaf9bdb5a3fb8a325115d9","abstract_canon_sha256":"2e4ec6e33c5241f849350ddb37648e3e8de95904dd331303332df5e9f6bc3f43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:46:37.659745Z","signature_b64":"YTiMz7K6eTqolK0ozsR3KMGVX0vwKThs/nKJ3ktUPhr+mpJERWBnMJd8AeIbVC4dlL4c+ovjEypXh9YK+joPDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80e93248acb1a0d84bed8b7a22633ea9a6116b48348c19b954f01c30b5ae42a9","last_reissued_at":"2026-07-05T09:46:37.659211Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:46:37.659211Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Lexicon: Rich Image Features in Language Space","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alireza Fathi, Cordelia Schmid, Trevor Darrell, Xingyi Zhou, Xudong Wang","submitted_at":"2024-12-09T18:57:24Z","abstract_excerpt":"We present Visual Lexicon, a novel visual language that encodes rich image information into the text space of vocabulary tokens while retaining intricate visual details that are often challenging to convey in natural language. Unlike traditional methods that prioritize either high-level semantics (e.g., CLIP) or pixel-level reconstruction (e.g., VAE), ViLex simultaneously captures rich semantic content and fine visual details, enabling high-quality image generation and comprehensive visual scene understanding. Through a self-supervised learning pipeline, ViLex generates tokens optimized for re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.06774","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.06774/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.06774","created_at":"2026-07-05T09:46:37.659272+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.06774v1","created_at":"2026-07-05T09:46:37.659272+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.06774","created_at":"2026-07-05T09:46:37.659272+00:00"},{"alias_kind":"pith_short_12","alias_value":"QDUTESFMWGQN","created_at":"2026-07-05T09:46:37.659272+00:00"},{"alias_kind":"pith_short_16","alias_value":"QDUTESFMWGQNQS7N","created_at":"2026-07-05T09:46:37.659272+00:00"},{"alias_kind":"pith_short_8","alias_value":"QDUTESFM","created_at":"2026-07-05T09:46:37.659272+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07104","citing_title":"Vision-Language-Vision Auto-Encoder: Scalable Knowledge Distillation from Diffusion Models","ref_index":66,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG","json":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG.json","graph_json":"https://pith.science/api/pith-number/QDUTESFMWGQNQS7NRN5CEYZ6VG/graph.json","events_json":"https://pith.science/api/pith-number/QDUTESFMWGQNQS7NRN5CEYZ6VG/events.json","paper":"https://pith.science/paper/QDUTESFM"},"agent_actions":{"view_html":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG","download_json":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG.json","view_paper":"https://pith.science/paper/QDUTESFM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.06774&json=true","fetch_graph":"https://pith.science/api/pith-number/QDUTESFMWGQNQS7NRN5CEYZ6VG/graph.json","fetch_events":"https://pith.science/api/pith-number/QDUTESFMWGQNQS7NRN5CEYZ6VG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG/action/storage_attestation","attest_author":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG/action/author_attestation","sign_citation":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG/action/citation_signature","submit_replication":"https://pith.science/pith/QDUTESFMWGQNQS7NRN5CEYZ6VG/action/replication_record"}},"created_at":"2026-07-05T09:46:37.659272+00:00","updated_at":"2026-07-05T09:46:37.659272+00:00"}