{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O6XXLB2CDM2YP5GZQSHTSPMX4T","short_pith_number":"pith:O6XXLB2C","schema_version":"1.0","canonical_sha256":"77af7587421b3587f4d9848f393d97e4ed91f817aeeae1355bac28e50e814da2","source":{"kind":"arxiv","id":"2408.13247","version":2},"attestation_state":"computed","paper":{"title":"An In-Depth Investigation of Data Collection in LLM App Ecosystems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.LG"],"primary_cat":"cs.CR","authors_text":"Evin Jaff, Ke Yang, Ning Zhang, Umar Iqbal, Yuhao Wu","submitted_at":"2024-08-23T17:42:06Z","abstract_excerpt":"LLM app (tool) ecosystems are rapidly evolving to support sophisticated use cases that often require extensive user data collection. Given that LLM apps are developed by third parties and anecdotal evidence indicating inconsistent enforcement of policies by LLM platforms, sharing user data with these apps presents significant privacy risks. In this paper, we aim to bring transparency in data practices of LLM app ecosystems. We examine OpenAI's GPT app ecosystem as a case study. We propose an LLM-based framework to analyze the natural language specifications of GPT Actions (custom tools) and as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.13247","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-08-23T17:42:06Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY","cs.LG"],"title_canon_sha256":"860a3f2ff904f181029fe22fae4ca8038f7390299d28217f860caf4c7351428a","abstract_canon_sha256":"63c75f696395322a2e67f59bdb7e5194a8dea6c7f7fad383de72e24439334b9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:54.323080Z","signature_b64":"/3ykinqT+SFU9dgZ8wMEHotpbMtj4Vh95CNW5mw35j786nVGZX/BdaE78YgnSgFoLPnC2gOzqkvkaXyWHqTaAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77af7587421b3587f4d9848f393d97e4ed91f817aeeae1355bac28e50e814da2","last_reissued_at":"2026-07-05T11:06:54.322567Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:54.322567Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An In-Depth Investigation of Data Collection in LLM App Ecosystems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.LG"],"primary_cat":"cs.CR","authors_text":"Evin Jaff, Ke Yang, Ning Zhang, Umar Iqbal, Yuhao Wu","submitted_at":"2024-08-23T17:42:06Z","abstract_excerpt":"LLM app (tool) ecosystems are rapidly evolving to support sophisticated use cases that often require extensive user data collection. Given that LLM apps are developed by third parties and anecdotal evidence indicating inconsistent enforcement of policies by LLM platforms, sharing user data with these apps presents significant privacy risks. In this paper, we aim to bring transparency in data practices of LLM app ecosystems. We examine OpenAI's GPT app ecosystem as a case study. We propose an LLM-based framework to analyze the natural language specifications of GPT Actions (custom tools) and as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.13247","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.13247/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.13247","created_at":"2026-07-05T11:06:54.322628+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.13247v2","created_at":"2026-07-05T11:06:54.322628+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.13247","created_at":"2026-07-05T11:06:54.322628+00:00"},{"alias_kind":"pith_short_12","alias_value":"O6XXLB2CDM2Y","created_at":"2026-07-05T11:06:54.322628+00:00"},{"alias_kind":"pith_short_16","alias_value":"O6XXLB2CDM2YP5GZ","created_at":"2026-07-05T11:06:54.322628+00:00"},{"alias_kind":"pith_short_8","alias_value":"O6XXLB2C","created_at":"2026-07-05T11:06:54.322628+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02501","citing_title":"ECG Foundation Models and Medical LLMs for Agentic Cardiovascular Intelligence at the Edge: A Review and Outlook","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11770","citing_title":"Behavioral Integrity Verification for AI Agent Skills","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15415","citing_title":"HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T","json":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T.json","graph_json":"https://pith.science/api/pith-number/O6XXLB2CDM2YP5GZQSHTSPMX4T/graph.json","events_json":"https://pith.science/api/pith-number/O6XXLB2CDM2YP5GZQSHTSPMX4T/events.json","paper":"https://pith.science/paper/O6XXLB2C"},"agent_actions":{"view_html":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T","download_json":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T.json","view_paper":"https://pith.science/paper/O6XXLB2C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.13247&json=true","fetch_graph":"https://pith.science/api/pith-number/O6XXLB2CDM2YP5GZQSHTSPMX4T/graph.json","fetch_events":"https://pith.science/api/pith-number/O6XXLB2CDM2YP5GZQSHTSPMX4T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T/action/storage_attestation","attest_author":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T/action/author_attestation","sign_citation":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T/action/citation_signature","submit_replication":"https://pith.science/pith/O6XXLB2CDM2YP5GZQSHTSPMX4T/action/replication_record"}},"created_at":"2026-07-05T11:06:54.322628+00:00","updated_at":"2026-07-05T11:06:54.322628+00:00"}