{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:U2GUVOSSMNJK24ZXI72ZNRLZJI","short_pith_number":"pith:U2GUVOSS","schema_version":"1.0","canonical_sha256":"a68d4aba526352ad733747f596c5794a2a397f4594bb761aeb1312dc16831459","source":{"kind":"arxiv","id":"2406.14021","version":2},"attestation_state":"computed","paper":{"title":"HIGHT: Hierarchical Graph Tokenization for Molecule-Language Alignment","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","q-bio.QM"],"primary_cat":"cs.CL","authors_text":"James Cheng, Juzheng Zhang, Quanming Yao, Yatao Bian, Yongqiang Chen","submitted_at":"2024-06-20T06:37:35Z","abstract_excerpt":"Recently, there has been a surge of interest in extending the success of large language models (LLMs) from texts to molecules. Most existing approaches adopt a graph neural network to represent a molecule as a series of node tokens for molecule-language alignment, which, however, have overlooked the inherent hierarchical structures in molecules. Notably, higher-order molecular structures contain rich semantics of functional groups, which encode crucial biochemical functionalities of the molecules. We show that neglecting the hierarchical information in tokenization will lead to subpar molecule"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.14021","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-20T06:37:35Z","cross_cats_sorted":["cs.LG","q-bio.QM"],"title_canon_sha256":"9b25d0899a6a1e09a02bd60e4dcedf8176cc3dcffa5678290ccb4ae48185c993","abstract_canon_sha256":"098674cfe198c8b4342dbdc9f938ce6e179fd2fb218f995330a01e2db3d56960"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:49.647089Z","signature_b64":"Sd8v3FVvYWRamElDC5+HXsUo1cqkWw1IftbOc+3DN3VlAgDLzMrQS1f1yKXwJFGE779csVjrlS3QJsiVdOdOAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a68d4aba526352ad733747f596c5794a2a397f4594bb761aeb1312dc16831459","last_reissued_at":"2026-07-05T11:16:49.646577Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:49.646577Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HIGHT: Hierarchical Graph Tokenization for Molecule-Language Alignment","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","q-bio.QM"],"primary_cat":"cs.CL","authors_text":"James Cheng, Juzheng Zhang, Quanming Yao, Yatao Bian, Yongqiang Chen","submitted_at":"2024-06-20T06:37:35Z","abstract_excerpt":"Recently, there has been a surge of interest in extending the success of large language models (LLMs) from texts to molecules. Most existing approaches adopt a graph neural network to represent a molecule as a series of node tokens for molecule-language alignment, which, however, have overlooked the inherent hierarchical structures in molecules. Notably, higher-order molecular structures contain rich semantics of functional groups, which encode crucial biochemical functionalities of the molecules. We show that neglecting the hierarchical information in tokenization will lead to subpar molecule"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.14021","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.14021/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.14021","created_at":"2026-07-05T11:16:49.646636+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.14021v2","created_at":"2026-07-05T11:16:49.646636+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.14021","created_at":"2026-07-05T11:16:49.646636+00:00"},{"alias_kind":"pith_short_12","alias_value":"U2GUVOSSMNJK","created_at":"2026-07-05T11:16:49.646636+00:00"},{"alias_kind":"pith_short_16","alias_value":"U2GUVOSSMNJK24ZX","created_at":"2026-07-05T11:16:49.646636+00:00"},{"alias_kind":"pith_short_8","alias_value":"U2GUVOSS","created_at":"2026-07-05T11:16:49.646636+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01982","citing_title":"MolSight: A Graph-Aware Vision-Language Model for Unified Chemical Image Understanding","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12369","citing_title":"A Hierarchical Quantized Tokenization Framework for Task-Adaptive Graph Representation Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02082","citing_title":"FARM: Enhancing Molecular Representations with Functional Group Awareness","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI","json":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI.json","graph_json":"https://pith.science/api/pith-number/U2GUVOSSMNJK24ZXI72ZNRLZJI/graph.json","events_json":"https://pith.science/api/pith-number/U2GUVOSSMNJK24ZXI72ZNRLZJI/events.json","paper":"https://pith.science/paper/U2GUVOSS"},"agent_actions":{"view_html":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI","download_json":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI.json","view_paper":"https://pith.science/paper/U2GUVOSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.14021&json=true","fetch_graph":"https://pith.science/api/pith-number/U2GUVOSSMNJK24ZXI72ZNRLZJI/graph.json","fetch_events":"https://pith.science/api/pith-number/U2GUVOSSMNJK24ZXI72ZNRLZJI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI/action/storage_attestation","attest_author":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI/action/author_attestation","sign_citation":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI/action/citation_signature","submit_replication":"https://pith.science/pith/U2GUVOSSMNJK24ZXI72ZNRLZJI/action/replication_record"}},"created_at":"2026-07-05T11:16:49.646636+00:00","updated_at":"2026-07-05T11:16:49.646636+00:00"}