{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BGE75D3GLDUQPA6AJYSAVCT5PB","short_pith_number":"pith:BGE75D3G","schema_version":"1.0","canonical_sha256":"0989fe8f6658e90783c04e240a8a7d78516c708c1ff3d348d92660163ee18a86","source":{"kind":"arxiv","id":"2502.01436","version":3},"attestation_state":"computed","paper":{"title":"Towards Safer Chatbots: Automated Policy Compliance Evaluation of Custom GPTs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"David Rodriguez, Jose M. del Alamo, Jose Such, William Seymour","submitted_at":"2025-02-03T15:19:28Z","abstract_excerpt":"User-configured chatbots built on top of large language models are increasingly available through centralized marketplaces such as OpenAI's GPT Store. While these platforms enforce usage policies intended to prevent harmful or inappropriate behavior, the scale and opacity of customized chatbots make systematic policy enforcement challenging. As a result, policy-violating chatbots continue to remain publicly accessible despite existing review processes. This paper presents a fully automated method for evaluating the compliance of Custom GPTs with its marketplace usage policy using black-box int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01436","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-03T15:19:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"18fb0e2cae2010ab4d0e293f528de017b45f0af910f60a2ebd3fdfa10e36abb8","abstract_canon_sha256":"b0e92cb342fba3d786e0f061ce5147691202ba10ffc329e9d18ac5f8107f8390"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:21:58.960636Z","signature_b64":"cgeTPW5LEIGSWXtgVvHCO0EnTE/FZAZn+Wl4596Kao5d8xd9fbfMCFULlSpnIEOpsoWMNzLpDHndcDl2WjzTDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0989fe8f6658e90783c04e240a8a7d78516c708c1ff3d348d92660163ee18a86","last_reissued_at":"2026-07-14T01:21:58.959658Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:21:58.959658Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Safer Chatbots: Automated Policy Compliance Evaluation of Custom GPTs","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"David Rodriguez, Jose M. del Alamo, Jose Such, William Seymour","submitted_at":"2025-02-03T15:19:28Z","abstract_excerpt":"User-configured chatbots built on top of large language models are increasingly available through centralized marketplaces such as OpenAI's GPT Store. While these platforms enforce usage policies intended to prevent harmful or inappropriate behavior, the scale and opacity of customized chatbots make systematic policy enforcement challenging. As a result, policy-violating chatbots continue to remain publicly accessible despite existing review processes. This paper presents a fully automated method for evaluating the compliance of Custom GPTs with its marketplace usage policy using black-box int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01436","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01436/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01436","created_at":"2026-07-14T01:21:58.960108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01436v3","created_at":"2026-07-14T01:21:58.960108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01436","created_at":"2026-07-14T01:21:58.960108+00:00"},{"alias_kind":"pith_short_12","alias_value":"BGE75D3GLDUQ","created_at":"2026-07-14T01:21:58.960108+00:00"},{"alias_kind":"pith_short_16","alias_value":"BGE75D3GLDUQPA6A","created_at":"2026-07-14T01:21:58.960108+00:00"},{"alias_kind":"pith_short_8","alias_value":"BGE75D3G","created_at":"2026-07-14T01:21:58.960108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.04394","citing_title":"Beyond Single-Policy: Evaluating Composed Organization-Specific Policy Alignment in LLM Chatbots","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2605.20591","citing_title":"Do No Harm? Hallucination and Actor-Level Abuse in Web-Deployed Medical Large Language Models","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB","json":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB.json","graph_json":"https://pith.science/api/pith-number/BGE75D3GLDUQPA6AJYSAVCT5PB/graph.json","events_json":"https://pith.science/api/pith-number/BGE75D3GLDUQPA6AJYSAVCT5PB/events.json","paper":"https://pith.science/paper/BGE75D3G"},"agent_actions":{"view_html":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB","download_json":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB.json","view_paper":"https://pith.science/paper/BGE75D3G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01436&json=true","fetch_graph":"https://pith.science/api/pith-number/BGE75D3GLDUQPA6AJYSAVCT5PB/graph.json","fetch_events":"https://pith.science/api/pith-number/BGE75D3GLDUQPA6AJYSAVCT5PB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB/action/storage_attestation","attest_author":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB/action/author_attestation","sign_citation":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB/action/citation_signature","submit_replication":"https://pith.science/pith/BGE75D3GLDUQPA6AJYSAVCT5PB/action/replication_record"}},"created_at":"2026-07-14T01:21:58.960108+00:00","updated_at":"2026-07-14T01:21:58.960108+00:00"}