{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CDQGKENJQHT3SF2HORUF2LDGPT","short_pith_number":"pith:CDQGKENJ","schema_version":"1.0","canonical_sha256":"10e06511a981e7b9174774685d2c667ccb6c924c0b0073b47ddfdc4796a28f10","source":{"kind":"arxiv","id":"2501.16125","version":3},"attestation_state":"computed","paper":{"title":"SampleLLM: Optimizing Tabular Data Synthesis in Recommendations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Huifeng Guo, Jingtong Gao, Ruiming Tang, Xiangyang Li, Xiangyu Zhao, Xiaopeng Li, Yichao Wang, Zhaocheng Du","submitted_at":"2025-01-27T15:12:27Z","abstract_excerpt":"Tabular data synthesis is crucial in machine learning, yet existing general methods-primarily based on statistical or deep learning models-are highly data-dependent and often fall short in recommender systems. This limitation arises from their difficulty in capturing complex distributions and understanding feature relationships from sparse and limited data, along with their inability to grasp semantic feature relations. Recently, Large Language Models (LLMs) have shown potential in generating synthetic data samples through few-shot learning and semantic understanding. However, they often suffe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16125","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2025-01-27T15:12:27Z","cross_cats_sorted":[],"title_canon_sha256":"0af576c91ada513daf06ea0f82660b9b9ce18cc6f797b4f566954ecac657d637","abstract_canon_sha256":"67393e23140aa1a4f60039642a4bb1f97fbde25ebc696899a0a212b2a843f4ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:25.763131Z","signature_b64":"ld/dDuYbu7sHDRFj0POhnwcpqcZruZfqC3Ai9M9Y6MhFQU8MpecgRn9N3WWSuiks3b0DzoWAJMDK5Z0OJyY8Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"10e06511a981e7b9174774685d2c667ccb6c924c0b0073b47ddfdc4796a28f10","last_reissued_at":"2026-07-05T10:12:25.762629Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:25.762629Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SampleLLM: Optimizing Tabular Data Synthesis in Recommendations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Huifeng Guo, Jingtong Gao, Ruiming Tang, Xiangyang Li, Xiangyu Zhao, Xiaopeng Li, Yichao Wang, Zhaocheng Du","submitted_at":"2025-01-27T15:12:27Z","abstract_excerpt":"Tabular data synthesis is crucial in machine learning, yet existing general methods-primarily based on statistical or deep learning models-are highly data-dependent and often fall short in recommender systems. This limitation arises from their difficulty in capturing complex distributions and understanding feature relationships from sparse and limited data, along with their inability to grasp semantic feature relations. Recently, Large Language Models (LLMs) have shown potential in generating synthetic data samples through few-shot learning and semantic understanding. However, they often suffe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16125","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16125","created_at":"2026-07-05T10:12:25.762693+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16125v3","created_at":"2026-07-05T10:12:25.762693+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16125","created_at":"2026-07-05T10:12:25.762693+00:00"},{"alias_kind":"pith_short_12","alias_value":"CDQGKENJQHT3","created_at":"2026-07-05T10:12:25.762693+00:00"},{"alias_kind":"pith_short_16","alias_value":"CDQGKENJQHT3SF2H","created_at":"2026-07-05T10:12:25.762693+00:00"},{"alias_kind":"pith_short_8","alias_value":"CDQGKENJ","created_at":"2026-07-05T10:12:25.762693+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27865","citing_title":"From Bootstrapping to Sequence Modeling: A Unified Generative Framework for Personalized Landing-Page Modeling","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT","json":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT.json","graph_json":"https://pith.science/api/pith-number/CDQGKENJQHT3SF2HORUF2LDGPT/graph.json","events_json":"https://pith.science/api/pith-number/CDQGKENJQHT3SF2HORUF2LDGPT/events.json","paper":"https://pith.science/paper/CDQGKENJ"},"agent_actions":{"view_html":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT","download_json":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT.json","view_paper":"https://pith.science/paper/CDQGKENJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16125&json=true","fetch_graph":"https://pith.science/api/pith-number/CDQGKENJQHT3SF2HORUF2LDGPT/graph.json","fetch_events":"https://pith.science/api/pith-number/CDQGKENJQHT3SF2HORUF2LDGPT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT/action/storage_attestation","attest_author":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT/action/author_attestation","sign_citation":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT/action/citation_signature","submit_replication":"https://pith.science/pith/CDQGKENJQHT3SF2HORUF2LDGPT/action/replication_record"}},"created_at":"2026-07-05T10:12:25.762693+00:00","updated_at":"2026-07-05T10:12:25.762693+00:00"}