{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GIWIBWYHUHPHQPJWYEYHNLGRH2","short_pith_number":"pith:GIWIBWYH","schema_version":"1.0","canonical_sha256":"322c80db07a1de783d36c13076acd13e8a44a2313bd4a87f3c14e378d5824258","source":{"kind":"arxiv","id":"2305.02626","version":1},"attestation_state":"computed","paper":{"title":"\"Oops, Did I Just Say That?\" Testing and Repairing Unethical Suggestions of Large Language Models with Suggest-Critique-Reflect Process","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.SE","authors_text":"Ao Sun, Pingchuan Ma, Shuai Wang, Zongjie Li","submitted_at":"2023-05-04T08:00:32Z","abstract_excerpt":"As the popularity of large language models (LLMs) soars across various applications, ensuring their alignment with human values has become a paramount concern. In particular, given that LLMs have great potential to serve as general-purpose AI assistants in daily life, their subtly unethical suggestions become a serious and real concern. Tackling the challenge of automatically testing and repairing unethical suggestions is thus demanding.\n  This paper introduces the first framework for testing and repairing unethical suggestions made by LLMs. We first propose ETHICSSUITE, a test suite that pres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.02626","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SE","submitted_at":"2023-05-04T08:00:32Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"4f946c0763bf1099752721e29539a0041a3fb1bb878dd773c39a95251e56fef9","abstract_canon_sha256":"9df2e587b6fe81a8d991385a6107f42c63c7f05556324e1cea281844040a1f8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:07:03.467233Z","signature_b64":"yQk3MpEHghFKoBhgs2BPtzqNfM4Z9uDv95otMjCg84LUcMqoEXg/TjySskmufSTfzbS6Y9vZ7OQXNuwcBZ6zDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"322c80db07a1de783d36c13076acd13e8a44a2313bd4a87f3c14e378d5824258","last_reissued_at":"2026-07-05T06:07:03.466812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:07:03.466812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"\"Oops, Did I Just Say That?\" Testing and Repairing Unethical Suggestions of Large Language Models with Suggest-Critique-Reflect Process","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.SE","authors_text":"Ao Sun, Pingchuan Ma, Shuai Wang, Zongjie Li","submitted_at":"2023-05-04T08:00:32Z","abstract_excerpt":"As the popularity of large language models (LLMs) soars across various applications, ensuring their alignment with human values has become a paramount concern. In particular, given that LLMs have great potential to serve as general-purpose AI assistants in daily life, their subtly unethical suggestions become a serious and real concern. Tackling the challenge of automatically testing and repairing unethical suggestions is thus demanding.\n  This paper introduces the first framework for testing and repairing unethical suggestions made by LLMs. We first propose ETHICSSUITE, a test suite that pres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.02626","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.02626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.02626","created_at":"2026-07-05T06:07:03.466878+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.02626v1","created_at":"2026-07-05T06:07:03.466878+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.02626","created_at":"2026-07-05T06:07:03.466878+00:00"},{"alias_kind":"pith_short_12","alias_value":"GIWIBWYHUHPH","created_at":"2026-07-05T06:07:03.466878+00:00"},{"alias_kind":"pith_short_16","alias_value":"GIWIBWYHUHPHQPJW","created_at":"2026-07-05T06:07:03.466878+00:00"},{"alias_kind":"pith_short_8","alias_value":"GIWIBWYH","created_at":"2026-07-05T06:07:03.466878+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.19759","citing_title":"Moral Reasoning Across Languages: The Critical Role of Low-Resource Languages in LLMs","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2","json":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2.json","graph_json":"https://pith.science/api/pith-number/GIWIBWYHUHPHQPJWYEYHNLGRH2/graph.json","events_json":"https://pith.science/api/pith-number/GIWIBWYHUHPHQPJWYEYHNLGRH2/events.json","paper":"https://pith.science/paper/GIWIBWYH"},"agent_actions":{"view_html":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2","download_json":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2.json","view_paper":"https://pith.science/paper/GIWIBWYH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.02626&json=true","fetch_graph":"https://pith.science/api/pith-number/GIWIBWYHUHPHQPJWYEYHNLGRH2/graph.json","fetch_events":"https://pith.science/api/pith-number/GIWIBWYHUHPHQPJWYEYHNLGRH2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2/action/storage_attestation","attest_author":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2/action/author_attestation","sign_citation":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2/action/citation_signature","submit_replication":"https://pith.science/pith/GIWIBWYHUHPHQPJWYEYHNLGRH2/action/replication_record"}},"created_at":"2026-07-05T06:07:03.466878+00:00","updated_at":"2026-07-05T06:07:03.466878+00:00"}