{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AA2YSUITOGFKGXYBQPPNMCAEPS","short_pith_number":"pith:AA2YSUIT","schema_version":"1.0","canonical_sha256":"0035895113718aa35f0183ded608047c82a9b1d8b50b60088d690cf56433a90a","source":{"kind":"arxiv","id":"2503.00089","version":2},"attestation_state":"computed","paper":{"title":"Protein Structure Tokenization: Benchmarking and New Recipe","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"q-bio.QM","authors_text":"Huzefa Rangwala, Marcus Collins, Xinyu Yuan, Zichen Wang","submitted_at":"2025-02-28T15:14:33Z","abstract_excerpt":"Recent years have witnessed a surge in the development of protein structural tokenization methods, which chunk protein 3D structures into discrete or continuous representations. Structure tokenization enables the direct application of powerful techniques like language modeling for protein structures, and large multimodal models to integrate structures with protein sequences and functional texts. Despite the progress, the capabilities and limitations of these methods remain poorly understood due to the lack of a unified evaluation framework. We first introduce StructTokenBench, a framework that"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00089","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"q-bio.QM","submitted_at":"2025-02-28T15:14:33Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f1f29f464c851b9a7d3aec062d717ebef451f024e73618b4409db0d193fc98a0","abstract_canon_sha256":"8f0c88d384b9298a1e407d82ccc7a99848011e72dbe3b779dfdaaf9ca71d6d39"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:48.457348Z","signature_b64":"6gpdn4hHk+5wBMm8knXMX3qflwcxfCtmPzwB68fPk6B7dxFcfrSSlbWoiI9BIVsYzo3vUI4Hug6a8TYrsRLbCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0035895113718aa35f0183ded608047c82a9b1d8b50b60088d690cf56433a90a","last_reissued_at":"2026-07-05T11:26:48.456804Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:48.456804Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Protein Structure Tokenization: Benchmarking and New Recipe","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"q-bio.QM","authors_text":"Huzefa Rangwala, Marcus Collins, Xinyu Yuan, Zichen Wang","submitted_at":"2025-02-28T15:14:33Z","abstract_excerpt":"Recent years have witnessed a surge in the development of protein structural tokenization methods, which chunk protein 3D structures into discrete or continuous representations. Structure tokenization enables the direct application of powerful techniques like language modeling for protein structures, and large multimodal models to integrate structures with protein sequences and functional texts. Despite the progress, the capabilities and limitations of these methods remain poorly understood due to the lack of a unified evaluation framework. We first introduce StructTokenBench, a framework that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00089","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00089/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00089","created_at":"2026-07-05T11:26:48.456864+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00089v2","created_at":"2026-07-05T11:26:48.456864+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00089","created_at":"2026-07-05T11:26:48.456864+00:00"},{"alias_kind":"pith_short_12","alias_value":"AA2YSUITOGFK","created_at":"2026-07-05T11:26:48.456864+00:00"},{"alias_kind":"pith_short_16","alias_value":"AA2YSUITOGFKGXYB","created_at":"2026-07-05T11:26:48.456864+00:00"},{"alias_kind":"pith_short_8","alias_value":"AA2YSUIT","created_at":"2026-07-05T11:26:48.456864+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13789","citing_title":"ENSEMBITS: an alphabet of protein conformational ensembles","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13789","citing_title":"ENSEMBITS: an alphabet of protein conformational ensembles","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09981","citing_title":"Yeti: A compact protein structure tokenizer for reconstruction and multi-modal generation","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS","json":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS.json","graph_json":"https://pith.science/api/pith-number/AA2YSUITOGFKGXYBQPPNMCAEPS/graph.json","events_json":"https://pith.science/api/pith-number/AA2YSUITOGFKGXYBQPPNMCAEPS/events.json","paper":"https://pith.science/paper/AA2YSUIT"},"agent_actions":{"view_html":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS","download_json":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS.json","view_paper":"https://pith.science/paper/AA2YSUIT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00089&json=true","fetch_graph":"https://pith.science/api/pith-number/AA2YSUITOGFKGXYBQPPNMCAEPS/graph.json","fetch_events":"https://pith.science/api/pith-number/AA2YSUITOGFKGXYBQPPNMCAEPS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS/action/storage_attestation","attest_author":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS/action/author_attestation","sign_citation":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS/action/citation_signature","submit_replication":"https://pith.science/pith/AA2YSUITOGFKGXYBQPPNMCAEPS/action/replication_record"}},"created_at":"2026-07-05T11:26:48.456864+00:00","updated_at":"2026-07-05T11:26:48.456864+00:00"}