{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:A327T5X56FW2HPTIT26CHSAPE6","short_pith_number":"pith:A327T5X5","schema_version":"1.0","canonical_sha256":"06f5f9f6fdf16da3be689ebc23c80f27a153d9445aa4e6be9546e1083f3d62dc","source":{"kind":"arxiv","id":"2507.18504","version":2},"attestation_state":"computed","paper":{"title":"Not All Features Deserve Attention: Graph-Guided Dependency Learning for Tabular Data Generation with Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bardh Prenkaj, Gjergji Kasneci, Shuo Yang, Zheyu Zhang","submitted_at":"2025-07-24T15:22:27Z","abstract_excerpt":"Large Language Models (LLMs) have shown strong potential for tabular data generation by modeling textualized feature-value pairs. However, tabular data inherently exhibits sparse feature-level dependencies, where many feature interactions are structurally insignificant. This creates a fundamental mismatch as LLMs' self-attention mechanism inevitably distributes focus across all pairs, diluting attention on critical relationships, particularly in datasets with complex dependencies or semantically ambiguous features. To address this limitation, we propose GraDe (Graph-Guided Dependency Learning)"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.18504","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-24T15:22:27Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"aa02f386ecd0ff20aa5d0986667fa9b0cd5b3d07acda9a338f077b86a4f9e537","abstract_canon_sha256":"dd7c66d96bbabb2336a64bea43d33d733d56959718c6f0c3574d0b91f4f94631"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:06:49.070749Z","signature_b64":"WxOoNTnq4qYpo53FHbh/pV+mmet4KtvWyo0hvesnScyUgmTDe91dOdxmBqeMsOEUl3AmHJLGmL4LvTJEJrYVCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06f5f9f6fdf16da3be689ebc23c80f27a153d9445aa4e6be9546e1083f3d62dc","last_reissued_at":"2026-07-05T12:06:49.070284Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:06:49.070284Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not All Features Deserve Attention: Graph-Guided Dependency Learning for Tabular Data Generation with Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bardh Prenkaj, Gjergji Kasneci, Shuo Yang, Zheyu Zhang","submitted_at":"2025-07-24T15:22:27Z","abstract_excerpt":"Large Language Models (LLMs) have shown strong potential for tabular data generation by modeling textualized feature-value pairs. However, tabular data inherently exhibits sparse feature-level dependencies, where many feature interactions are structurally insignificant. This creates a fundamental mismatch as LLMs' self-attention mechanism inevitably distributes focus across all pairs, diluting attention on critical relationships, particularly in datasets with complex dependencies or semantically ambiguous features. To address this limitation, we propose GraDe (Graph-Guided Dependency Learning)"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.18504","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.18504/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.18504","created_at":"2026-07-05T12:06:49.070350+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.18504v2","created_at":"2026-07-05T12:06:49.070350+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.18504","created_at":"2026-07-05T12:06:49.070350+00:00"},{"alias_kind":"pith_short_12","alias_value":"A327T5X56FW2","created_at":"2026-07-05T12:06:49.070350+00:00"},{"alias_kind":"pith_short_16","alias_value":"A327T5X56FW2HPTI","created_at":"2026-07-05T12:06:49.070350+00:00"},{"alias_kind":"pith_short_8","alias_value":"A327T5X5","created_at":"2026-07-05T12:06:49.070350+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.02040","citing_title":"Attributes as Textual Genes: Leveraging LLMs as Genetic Algorithm Simulators for Conditional Synthetic Data Generation","ref_index":5,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6","json":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6.json","graph_json":"https://pith.science/api/pith-number/A327T5X56FW2HPTIT26CHSAPE6/graph.json","events_json":"https://pith.science/api/pith-number/A327T5X56FW2HPTIT26CHSAPE6/events.json","paper":"https://pith.science/paper/A327T5X5"},"agent_actions":{"view_html":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6","download_json":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6.json","view_paper":"https://pith.science/paper/A327T5X5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.18504&json=true","fetch_graph":"https://pith.science/api/pith-number/A327T5X56FW2HPTIT26CHSAPE6/graph.json","fetch_events":"https://pith.science/api/pith-number/A327T5X56FW2HPTIT26CHSAPE6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6/action/storage_attestation","attest_author":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6/action/author_attestation","sign_citation":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6/action/citation_signature","submit_replication":"https://pith.science/pith/A327T5X56FW2HPTIT26CHSAPE6/action/replication_record"}},"created_at":"2026-07-05T12:06:49.070350+00:00","updated_at":"2026-07-05T12:06:49.070350+00:00"}