{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RYCXIYM35NRWX565QJB3I3TZRQ","short_pith_number":"pith:RYCXIYM3","schema_version":"1.0","canonical_sha256":"8e0574619beb636bf7dd8243b46e798c19bd2f11c7654a5d5894e75d1548bc0f","source":{"kind":"arxiv","id":"2508.03766","version":1},"attestation_state":"computed","paper":{"title":"LLM-Prior: A Framework for Knowledge-Driven Prior Elicitation and Aggregation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Yongchao Huang","submitted_at":"2025-08-05T01:43:29Z","abstract_excerpt":"The specification of prior distributions is fundamental in Bayesian inference, yet it remains a significant bottleneck. The prior elicitation process is often a manual, subjective, and unscalable task. We propose a novel framework which leverages Large Language Models (LLMs) to automate and scale this process. We introduce \\texttt{LLMPrior}, a principled operator that translates rich, unstructured contexts such as natural language descriptions, data or figures into valid, tractable probability distributions. We formalize this operator by architecturally coupling an LLM with an explicit, tracta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.03766","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-05T01:43:29Z","cross_cats_sorted":[],"title_canon_sha256":"a85cdf528f16e17a43fcbb209d2bb224e2e2999a85c2a91565307991e7e97c5a","abstract_canon_sha256":"2f2586a3c23e8fa1d9af29ee0bd832b4adff173d75a8363985c99d4f02f53a45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:14.780077Z","signature_b64":"S8otRIX2buISmQW7hqtt64T1ELvkMohJvI2ReolOg3R50FE9EhahHG+/82G3RB6k9bYADRkLZhfvpChLvHgEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8e0574619beb636bf7dd8243b46e798c19bd2f11c7654a5d5894e75d1548bc0f","last_reissued_at":"2026-07-05T11:49:14.779662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:14.779662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM-Prior: A Framework for Knowledge-Driven Prior Elicitation and Aggregation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Yongchao Huang","submitted_at":"2025-08-05T01:43:29Z","abstract_excerpt":"The specification of prior distributions is fundamental in Bayesian inference, yet it remains a significant bottleneck. The prior elicitation process is often a manual, subjective, and unscalable task. We propose a novel framework which leverages Large Language Models (LLMs) to automate and scale this process. We introduce \\texttt{LLMPrior}, a principled operator that translates rich, unstructured contexts such as natural language descriptions, data or figures into valid, tractable probability distributions. We formalize this operator by architecturally coupling an LLM with an explicit, tracta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.03766","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.03766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.03766","created_at":"2026-07-05T11:49:14.779723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.03766v1","created_at":"2026-07-05T11:49:14.779723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.03766","created_at":"2026-07-05T11:49:14.779723+00:00"},{"alias_kind":"pith_short_12","alias_value":"RYCXIYM35NRW","created_at":"2026-07-05T11:49:14.779723+00:00"},{"alias_kind":"pith_short_16","alias_value":"RYCXIYM35NRWX565","created_at":"2026-07-05T11:49:14.779723+00:00"},{"alias_kind":"pith_short_8","alias_value":"RYCXIYM3","created_at":"2026-07-05T11:49:14.779723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31630","citing_title":"Calibration, Not Compilation: Detecting and Repairing Misspecified Probabilistic Programs Written by Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18476","citing_title":"AI4BayesCode: From Natural Language Descriptions to Validated Modular Stateful Bayesian Samplers","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ","json":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ.json","graph_json":"https://pith.science/api/pith-number/RYCXIYM35NRWX565QJB3I3TZRQ/graph.json","events_json":"https://pith.science/api/pith-number/RYCXIYM35NRWX565QJB3I3TZRQ/events.json","paper":"https://pith.science/paper/RYCXIYM3"},"agent_actions":{"view_html":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ","download_json":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ.json","view_paper":"https://pith.science/paper/RYCXIYM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.03766&json=true","fetch_graph":"https://pith.science/api/pith-number/RYCXIYM35NRWX565QJB3I3TZRQ/graph.json","fetch_events":"https://pith.science/api/pith-number/RYCXIYM35NRWX565QJB3I3TZRQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ/action/storage_attestation","attest_author":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ/action/author_attestation","sign_citation":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ/action/citation_signature","submit_replication":"https://pith.science/pith/RYCXIYM35NRWX565QJB3I3TZRQ/action/replication_record"}},"created_at":"2026-07-05T11:49:14.779723+00:00","updated_at":"2026-07-05T11:49:14.779723+00:00"}