{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JGZD5L3ZIJRMUXNNU65HNBSYVE","short_pith_number":"pith:JGZD5L3Z","schema_version":"1.0","canonical_sha256":"49b23eaf794262ca5dada7ba768658a934124fa64f91f0d42b02c76fd44c7751","source":{"kind":"arxiv","id":"2402.05862","version":1},"attestation_state":"computed","paper":{"title":"Let Your Graph Do the Talking: Encoding Structured Data for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anton Tsitsulin, Bahare Fatemi, Bryan Perozzi, Dustin Zelle, Jonathan Halcrow, Mehran Kazemi, Rami Al-Rfou","submitted_at":"2024-02-08T17:51:44Z","abstract_excerpt":"How can we best encode structured data into sequential form for use in large language models (LLMs)? In this work, we introduce a parameter-efficient method to explicitly represent structured data for LLMs. Our method, GraphToken, learns an encoding function to extend prompts with explicit structured information. Unlike other work which focuses on limited domains (e.g. knowledge graph representation), our work is the first effort focused on the general encoding of structured data to be used for various reasoning tasks. We show that explicitly representing the graph structure allows significant"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05862","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-08T17:51:44Z","cross_cats_sorted":["cs.AI","cs.SI","stat.ML"],"title_canon_sha256":"40df187f5a3f87e954e9ebe819422fb9594b1c5f46e7730838d9554c55e696eb","abstract_canon_sha256":"2a9fbaee99d4fb82efc0ac0e1684c86fe188c662c38e15056b3449a9c7a01121"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:43:04.367835Z","signature_b64":"0fviZJN86zQSgE8pcishGkLxJNCPlf1L+rC3eIVZWvosf4nxVercHbzs1CV4crADaE/Zm7j7v8Z9U9znAWe4BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49b23eaf794262ca5dada7ba768658a934124fa64f91f0d42b02c76fd44c7751","last_reissued_at":"2026-07-05T07:43:04.367426Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:43:04.367426Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Let Your Graph Do the Talking: Encoding Structured Data for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anton Tsitsulin, Bahare Fatemi, Bryan Perozzi, Dustin Zelle, Jonathan Halcrow, Mehran Kazemi, Rami Al-Rfou","submitted_at":"2024-02-08T17:51:44Z","abstract_excerpt":"How can we best encode structured data into sequential form for use in large language models (LLMs)? In this work, we introduce a parameter-efficient method to explicitly represent structured data for LLMs. Our method, GraphToken, learns an encoding function to extend prompts with explicit structured information. Unlike other work which focuses on limited domains (e.g. knowledge graph representation), our work is the first effort focused on the general encoding of structured data to be used for various reasoning tasks. We show that explicitly representing the graph structure allows significant"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05862","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05862/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05862","created_at":"2026-07-05T07:43:04.367483+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05862v1","created_at":"2026-07-05T07:43:04.367483+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05862","created_at":"2026-07-05T07:43:04.367483+00:00"},{"alias_kind":"pith_short_12","alias_value":"JGZD5L3ZIJRM","created_at":"2026-07-05T07:43:04.367483+00:00"},{"alias_kind":"pith_short_16","alias_value":"JGZD5L3ZIJRMUXNN","created_at":"2026-07-05T07:43:04.367483+00:00"},{"alias_kind":"pith_short_8","alias_value":"JGZD5L3Z","created_at":"2026-07-05T07:43:04.367483+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11562","citing_title":"GraphInfer-Bench: Benchmarking LLM's Inference Capability on Graphs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06865","citing_title":"Are Large Language Models Suitable for Graph Computation? Progress and Prospects","ref_index":227,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06073","citing_title":"Edge-Aware Curvature Modeling for Graph Understanding in Large Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28175","citing_title":"Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2501.17549","citing_title":"Query-Aware Learnable Graph Pooling Tokens as Prompt for Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20170","citing_title":"KoRe: Compact Knowledge Representations for Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14785","citing_title":"Exploring the In-Context Learning Capabilities of LLMs for Money Laundering Detection in Financial Graphs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12197","citing_title":"A Unified Graph Language Model for Multi-Domain Multi-Task Graph Alignment Instruction Tuning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12061","citing_title":"SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04834","citing_title":"Bridging Input Feature Spaces Towards Graph Foundation Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07840","citing_title":"RelAgent: LLM Agents as Data Scientists for Relational Learning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27351","citing_title":"Heterogeneous Scientific Foundation Model Collaboration","ref_index":143,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE","json":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE.json","graph_json":"https://pith.science/api/pith-number/JGZD5L3ZIJRMUXNNU65HNBSYVE/graph.json","events_json":"https://pith.science/api/pith-number/JGZD5L3ZIJRMUXNNU65HNBSYVE/events.json","paper":"https://pith.science/paper/JGZD5L3Z"},"agent_actions":{"view_html":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE","download_json":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE.json","view_paper":"https://pith.science/paper/JGZD5L3Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05862&json=true","fetch_graph":"https://pith.science/api/pith-number/JGZD5L3ZIJRMUXNNU65HNBSYVE/graph.json","fetch_events":"https://pith.science/api/pith-number/JGZD5L3ZIJRMUXNNU65HNBSYVE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE/action/storage_attestation","attest_author":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE/action/author_attestation","sign_citation":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE/action/citation_signature","submit_replication":"https://pith.science/pith/JGZD5L3ZIJRMUXNNU65HNBSYVE/action/replication_record"}},"created_at":"2026-07-05T07:43:04.367483+00:00","updated_at":"2026-07-05T07:43:04.367483+00:00"}