{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KDSL4PVOKZ2M253XOJMKQCR5WD","short_pith_number":"pith:KDSL4PVO","schema_version":"1.0","canonical_sha256":"50e4be3eae5674cd77777258a80a3db0f76b6a7178738e5a63987ee5f6b5af43","source":{"kind":"arxiv","id":"2505.23842","version":5},"attestation_state":"computed","paper":{"title":"Fair Document Valuation in LLM Summaries via Shapley Values","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["econ.GN","q-fin.EC"],"primary_cat":"cs.CL","authors_text":"Hema Yoganarasimhan, Zikun Ye","submitted_at":"2025-05-28T15:14:21Z","abstract_excerpt":"Large Language Models (LLMs) increasingly power search engines and AI assistants that retrieve and summarize content from many sources. By serving answers directly, these systems obscure the original content creators' contributions, threatening the compensation that sustains a healthy content ecosystem. We frame this as a problem of fair document valuation and compensation, and propose a framework based on the Shapley value. Because exact Shapley computation is prohibitively expensive at scale, we develop Cluster Shapley, an approximation that groups semantically similar documents via LLM embe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23842","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T15:14:21Z","cross_cats_sorted":["econ.GN","q-fin.EC"],"title_canon_sha256":"6fab148436e4cd9d3337b5a9918f3457965a094c48e8ac39a987bd86c7ed4d94","abstract_canon_sha256":"2920d85873fc6309e82e78cd9ae57ad3ebffcb12f265c770f474db7898a6d65c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-10T00:18:40.855641Z","signature_b64":"zMUmaq73ECLLuWXNHWmucuSB52lcVeIfG3MUSGww6QR7gY5q/ynd8/fQRwfOow1r5zq8Ehns62PT4v7AT8I2Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50e4be3eae5674cd77777258a80a3db0f76b6a7178738e5a63987ee5f6b5af43","last_reissued_at":"2026-07-10T00:18:40.855113Z","signature_status":"signed_v1","first_computed_at":"2026-07-10T00:18:40.855113Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fair Document Valuation in LLM Summaries via Shapley Values","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["econ.GN","q-fin.EC"],"primary_cat":"cs.CL","authors_text":"Hema Yoganarasimhan, Zikun Ye","submitted_at":"2025-05-28T15:14:21Z","abstract_excerpt":"Large Language Models (LLMs) increasingly power search engines and AI assistants that retrieve and summarize content from many sources. By serving answers directly, these systems obscure the original content creators' contributions, threatening the compensation that sustains a healthy content ecosystem. We frame this as a problem of fair document valuation and compensation, and propose a framework based on the Shapley value. Because exact Shapley computation is prohibitively expensive at scale, we develop Cluster Shapley, an approximation that groups semantically similar documents via LLM embe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23842","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23842/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23842","created_at":"2026-07-10T00:18:40.855173+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23842v5","created_at":"2026-07-10T00:18:40.855173+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23842","created_at":"2026-07-10T00:18:40.855173+00:00"},{"alias_kind":"pith_short_12","alias_value":"KDSL4PVOKZ2M","created_at":"2026-07-10T00:18:40.855173+00:00"},{"alias_kind":"pith_short_16","alias_value":"KDSL4PVOKZ2M253X","created_at":"2026-07-10T00:18:40.855173+00:00"},{"alias_kind":"pith_short_8","alias_value":"KDSL4PVO","created_at":"2026-07-10T00:18:40.855173+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07529","citing_title":"FedMark-FM: Auditable, Risk-Adjusted Data Markets for Federated Foundation-Model Adaptation","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD","json":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD.json","graph_json":"https://pith.science/api/pith-number/KDSL4PVOKZ2M253XOJMKQCR5WD/graph.json","events_json":"https://pith.science/api/pith-number/KDSL4PVOKZ2M253XOJMKQCR5WD/events.json","paper":"https://pith.science/paper/KDSL4PVO"},"agent_actions":{"view_html":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD","download_json":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD.json","view_paper":"https://pith.science/paper/KDSL4PVO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23842&json=true","fetch_graph":"https://pith.science/api/pith-number/KDSL4PVOKZ2M253XOJMKQCR5WD/graph.json","fetch_events":"https://pith.science/api/pith-number/KDSL4PVOKZ2M253XOJMKQCR5WD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD/action/storage_attestation","attest_author":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD/action/author_attestation","sign_citation":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD/action/citation_signature","submit_replication":"https://pith.science/pith/KDSL4PVOKZ2M253XOJMKQCR5WD/action/replication_record"}},"created_at":"2026-07-10T00:18:40.855173+00:00","updated_at":"2026-07-10T00:18:40.855173+00:00"}