{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:673C3JTGBQGO5R3NL5FHF2XSKW","short_pith_number":"pith:673C3JTG","schema_version":"1.0","canonical_sha256":"f7f62da6660c0ceec76d5f4a72eaf255968fa5dd2439f20550aaeb0552554b2c","source":{"kind":"arxiv","id":"2505.04196","version":1},"attestation_state":"computed","paper":{"title":"A Large Language Model for Feasible and Diverse Population Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Dong-Kyu Kim, Eui-Jin Kim, Hyunsoo Yun, Prateek Bansal, Sung Yoo Lim","submitted_at":"2025-05-07T07:50:12Z","abstract_excerpt":"Generating a synthetic population that is both feasible and diverse is crucial for ensuring the validity of downstream activity schedule simulation in activity-based models (ABMs). While deep generative models (DGMs), such as variational autoencoders and generative adversarial networks, have been applied to this task, they often struggle to balance the inclusion of rare but plausible combinations (i.e., sampling zeros) with the exclusion of implausible ones (i.e., structural zeros). To improve feasibility while maintaining diversity, we propose a fine-tuning method for large language models (L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.04196","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-07T07:50:12Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"e9f00e8271839bb809567db6c36bf563c8156ac60a81a1684bdabcf36f4339dc","abstract_canon_sha256":"93c1b591727b5072789f8d0c05d826b81dce842f3541a2082c6cbfbf6d6a6e0f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:38.169146Z","signature_b64":"D0BQ6OR703I6UHJgualcKkCyJm7eYWg7WcJUWrjDoKXXvlnnflaVRwWeX72mGpSihSLxQUKr1jY2NBRPUlbSDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7f62da6660c0ceec76d5f4a72eaf255968fa5dd2439f20550aaeb0552554b2c","last_reissued_at":"2026-07-05T10:59:38.168641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:38.168641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Large Language Model for Feasible and Diverse Population Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.LG","authors_text":"Dong-Kyu Kim, Eui-Jin Kim, Hyunsoo Yun, Prateek Bansal, Sung Yoo Lim","submitted_at":"2025-05-07T07:50:12Z","abstract_excerpt":"Generating a synthetic population that is both feasible and diverse is crucial for ensuring the validity of downstream activity schedule simulation in activity-based models (ABMs). While deep generative models (DGMs), such as variational autoencoders and generative adversarial networks, have been applied to this task, they often struggle to balance the inclusion of rare but plausible combinations (i.e., sampling zeros) with the exclusion of implausible ones (i.e., structural zeros). To improve feasibility while maintaining diversity, we propose a fine-tuning method for large language models (L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.04196","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.04196/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.04196","created_at":"2026-07-05T10:59:38.168705+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.04196v1","created_at":"2026-07-05T10:59:38.168705+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.04196","created_at":"2026-07-05T10:59:38.168705+00:00"},{"alias_kind":"pith_short_12","alias_value":"673C3JTGBQGO","created_at":"2026-07-05T10:59:38.168705+00:00"},{"alias_kind":"pith_short_16","alias_value":"673C3JTGBQGO5R3N","created_at":"2026-07-05T10:59:38.168705+00:00"},{"alias_kind":"pith_short_8","alias_value":"673C3JTG","created_at":"2026-07-05T10:59:38.168705+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27650","citing_title":"GenWorld: Empirically Grounded Urban Simulation Infrastructure for Scalable LLM-Agent Studies","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31489","citing_title":"Context-Conditioned Generative Models Enable Subnational Refinement of Sparse Humanitarian Surveys","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW","json":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW.json","graph_json":"https://pith.science/api/pith-number/673C3JTGBQGO5R3NL5FHF2XSKW/graph.json","events_json":"https://pith.science/api/pith-number/673C3JTGBQGO5R3NL5FHF2XSKW/events.json","paper":"https://pith.science/paper/673C3JTG"},"agent_actions":{"view_html":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW","download_json":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW.json","view_paper":"https://pith.science/paper/673C3JTG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.04196&json=true","fetch_graph":"https://pith.science/api/pith-number/673C3JTGBQGO5R3NL5FHF2XSKW/graph.json","fetch_events":"https://pith.science/api/pith-number/673C3JTGBQGO5R3NL5FHF2XSKW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW/action/storage_attestation","attest_author":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW/action/author_attestation","sign_citation":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW/action/citation_signature","submit_replication":"https://pith.science/pith/673C3JTGBQGO5R3NL5FHF2XSKW/action/replication_record"}},"created_at":"2026-07-05T10:59:38.168705+00:00","updated_at":"2026-07-05T10:59:38.168705+00:00"}