{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6TSNVTLRLD25DPTRRQ7FSIW3IF","short_pith_number":"pith:6TSNVTLR","schema_version":"1.0","canonical_sha256":"f4e4dacd7158f5d1be718c3e5922db417dd0b03780f8e08c0a4a5da8c4bfd94f","source":{"kind":"arxiv","id":"2412.18702","version":2},"attestation_state":"computed","paper":{"title":"CypherBench: Towards Precise Retrieval over Full-scale Modern Knowledge Graphs in the LLM Era","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DB"],"primary_cat":"cs.CL","authors_text":"Sajjadur Rahman, Simone Papicchio, Yanlin Feng","submitted_at":"2024-12-24T23:22:04Z","abstract_excerpt":"Retrieval from graph data is crucial for augmenting large language models (LLM) with both open-domain knowledge and private enterprise data, and it is also a key component in the recent GraphRAG system (edge et al., 2024). Despite decades of research on knowledge graphs and knowledge base question answering, leading LLM frameworks (e.g. Langchain and LlamaIndex) have only minimal support for retrieval from modern encyclopedic knowledge graphs like Wikidata. In this paper, we analyze the root cause and suggest that modern RDF knowledge graphs (e.g. Wikidata, Freebase) are less efficient for LLM"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18702","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-24T23:22:04Z","cross_cats_sorted":["cs.AI","cs.DB"],"title_canon_sha256":"0c7ca359be6437f1315c1a4b0540381947f203715478e3d579dfa07c82ae7cb0","abstract_canon_sha256":"928abefe6758d24266b7d73844c1a13491907e0b7570826f79370fb2fe45873d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:42.192870Z","signature_b64":"XzO7GdqLY5U4e3H1qI9hSj4m9F8TSMSZUo8ByZ0z9mXA0XHHY0zO5TGqoKU49ecgbolKXQsPncT8kEynsLvBCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4e4dacd7158f5d1be718c3e5922db417dd0b03780f8e08c0a4a5da8c4bfd94f","last_reissued_at":"2026-07-05T10:44:42.192374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:42.192374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CypherBench: Towards Precise Retrieval over Full-scale Modern Knowledge Graphs in the LLM Era","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DB"],"primary_cat":"cs.CL","authors_text":"Sajjadur Rahman, Simone Papicchio, Yanlin Feng","submitted_at":"2024-12-24T23:22:04Z","abstract_excerpt":"Retrieval from graph data is crucial for augmenting large language models (LLM) with both open-domain knowledge and private enterprise data, and it is also a key component in the recent GraphRAG system (edge et al., 2024). Despite decades of research on knowledge graphs and knowledge base question answering, leading LLM frameworks (e.g. Langchain and LlamaIndex) have only minimal support for retrieval from modern encyclopedic knowledge graphs like Wikidata. In this paper, we analyze the root cause and suggest that modern RDF knowledge graphs (e.g. Wikidata, Freebase) are less efficient for LLM"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18702","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18702/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18702","created_at":"2026-07-05T10:44:42.192434+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18702v2","created_at":"2026-07-05T10:44:42.192434+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18702","created_at":"2026-07-05T10:44:42.192434+00:00"},{"alias_kind":"pith_short_12","alias_value":"6TSNVTLRLD25","created_at":"2026-07-05T10:44:42.192434+00:00"},{"alias_kind":"pith_short_16","alias_value":"6TSNVTLRLD25DPTR","created_at":"2026-07-05T10:44:42.192434+00:00"},{"alias_kind":"pith_short_8","alias_value":"6TSNVTLR","created_at":"2026-07-05T10:44:42.192434+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08481","citing_title":"PIPE-Cypher: Automatic Enterprise Benchmark Generation for Text-to-Cypher Systems","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00845","citing_title":"Graph Query Generation with Constraint-guided Large Language Agents","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF","json":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF.json","graph_json":"https://pith.science/api/pith-number/6TSNVTLRLD25DPTRRQ7FSIW3IF/graph.json","events_json":"https://pith.science/api/pith-number/6TSNVTLRLD25DPTRRQ7FSIW3IF/events.json","paper":"https://pith.science/paper/6TSNVTLR"},"agent_actions":{"view_html":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF","download_json":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF.json","view_paper":"https://pith.science/paper/6TSNVTLR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18702&json=true","fetch_graph":"https://pith.science/api/pith-number/6TSNVTLRLD25DPTRRQ7FSIW3IF/graph.json","fetch_events":"https://pith.science/api/pith-number/6TSNVTLRLD25DPTRRQ7FSIW3IF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF/action/storage_attestation","attest_author":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF/action/author_attestation","sign_citation":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF/action/citation_signature","submit_replication":"https://pith.science/pith/6TSNVTLRLD25DPTRRQ7FSIW3IF/action/replication_record"}},"created_at":"2026-07-05T10:44:42.192434+00:00","updated_at":"2026-07-05T10:44:42.192434+00:00"}