{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FHK7AUTHA37B3HGAZS6MEZYJIR","short_pith_number":"pith:FHK7AUTH","schema_version":"1.0","canonical_sha256":"29d5f0526706fe1d9cc0ccbcc26709446fccf11ec4f2af9b0232a2f7f2b7b17a","source":{"kind":"arxiv","id":"2411.05001","version":1},"attestation_state":"computed","paper":{"title":"Analyzing The Language of Visual Tokens","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Cheol Jun Cho, David M. Chan, Joonyong Park, Rodolfo Corona, Trevor Darrell, Yutong Bai","submitted_at":"2024-11-07T18:59:28Z","abstract_excerpt":"With the introduction of transformer-based models for vision and language tasks, such as LLaVA and Chameleon, there has been renewed interest in the discrete tokenized representation of images. These models often treat image patches as discrete tokens, analogous to words in natural language, learning joint alignments between visual and human languages. However, little is known about the statistical behavior of these visual languages - whether they follow similar frequency distributions, grammatical structures, or topologies as natural languages. In this paper, we take a natural-language-centri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.05001","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-07T18:59:28Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"59f3edf11d74caeae8355d6a47763fa43b7f6b1effeec754d4775208cd697d9d","abstract_canon_sha256":"3b76fca5880a3af4ee4d0fd3ab303dc0343c71eb8f3a5aef6837847f24609f59"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:37.447836Z","signature_b64":"D25aZSsS0CWmz+hDeN9Jwg0wvdcJsWFlnUFFvnM2pYGvVMZ9CrTS63RxQlJxFSCEFco6kkml2/eq6OVLMvfABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29d5f0526706fe1d9cc0ccbcc26709446fccf11ec4f2af9b0232a2f7f2b7b17a","last_reissued_at":"2026-07-05T09:32:37.447347Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:37.447347Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing The Language of Visual Tokens","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Cheol Jun Cho, David M. Chan, Joonyong Park, Rodolfo Corona, Trevor Darrell, Yutong Bai","submitted_at":"2024-11-07T18:59:28Z","abstract_excerpt":"With the introduction of transformer-based models for vision and language tasks, such as LLaVA and Chameleon, there has been renewed interest in the discrete tokenized representation of images. These models often treat image patches as discrete tokens, analogous to words in natural language, learning joint alignments between visual and human languages. However, little is known about the statistical behavior of these visual languages - whether they follow similar frequency distributions, grammatical structures, or topologies as natural languages. In this paper, we take a natural-language-centri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.05001","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.05001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.05001","created_at":"2026-07-05T09:32:37.447406+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.05001v1","created_at":"2026-07-05T09:32:37.447406+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.05001","created_at":"2026-07-05T09:32:37.447406+00:00"},{"alias_kind":"pith_short_12","alias_value":"FHK7AUTHA37B","created_at":"2026-07-05T09:32:37.447406+00:00"},{"alias_kind":"pith_short_16","alias_value":"FHK7AUTHA37B3HGA","created_at":"2026-07-05T09:32:37.447406+00:00"},{"alias_kind":"pith_short_8","alias_value":"FHK7AUTH","created_at":"2026-07-05T09:32:37.447406+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28643","citing_title":"Obliviate: Erasing Concepts from Autoregressive Image Generation Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19227","citing_title":"Token by Token, Compromised: Backdoor Vulnerabilities in Unified Autoregressive Models","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR","json":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR.json","graph_json":"https://pith.science/api/pith-number/FHK7AUTHA37B3HGAZS6MEZYJIR/graph.json","events_json":"https://pith.science/api/pith-number/FHK7AUTHA37B3HGAZS6MEZYJIR/events.json","paper":"https://pith.science/paper/FHK7AUTH"},"agent_actions":{"view_html":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR","download_json":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR.json","view_paper":"https://pith.science/paper/FHK7AUTH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.05001&json=true","fetch_graph":"https://pith.science/api/pith-number/FHK7AUTHA37B3HGAZS6MEZYJIR/graph.json","fetch_events":"https://pith.science/api/pith-number/FHK7AUTHA37B3HGAZS6MEZYJIR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR/action/storage_attestation","attest_author":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR/action/author_attestation","sign_citation":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR/action/citation_signature","submit_replication":"https://pith.science/pith/FHK7AUTHA37B3HGAZS6MEZYJIR/action/replication_record"}},"created_at":"2026-07-05T09:32:37.447406+00:00","updated_at":"2026-07-05T09:32:37.447406+00:00"}