{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5RC57MVHXYFM2UWJVVXEBE4UGZ","short_pith_number":"pith:5RC57MVH","schema_version":"1.0","canonical_sha256":"ec45dfb2a7be0acd52c9ad6e40939436477eba652faf35650e417beae00b704b","source":{"kind":"arxiv","id":"2311.04235","version":3},"attestation_state":"computed","paper":{"title":"Can LLMs Follow Simple Rules?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Basel Alomair, Dan Hendrycks, David Karamardian, David Wagner, Lulwa Aljeraisy, Norman Mu, Sarah Chen, Sizhe Chen, Zifan Wang","submitted_at":"2023-11-06T08:50:29Z","abstract_excerpt":"As Large Language Models (LLMs) are deployed with increasing real-world responsibilities, it is important to be able to specify and constrain the behavior of these systems in a reliable manner. Model developers may wish to set explicit rules for the model, such as \"do not generate abusive content\", but these may be circumvented by jailbreaking techniques. Existing evaluations of adversarial attacks and defenses on LLMs generally require either expensive manual review or unreliable heuristic checks. To address this issue, we propose Rule-following Language Evaluation Scenarios (RuLES), a progra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.04235","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-11-06T08:50:29Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"0d3ecb9f59edc928b360b39735e66e4c78e8520115e1e4f6ca225882f97e2df6","abstract_canon_sha256":"7ea6a1a247193d32b7cc09a5ffca79e5fe5ab842d7c322b2af1d13f10c037a8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:48.939769Z","signature_b64":"4M/DXwVeJSpAJdRxFyB/a9PAtb+X9KzeM1E1ClknDSUoSDSJsiD22hY0EAL36sMKIJdJ2I5rgDQVshLJNQWcDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec45dfb2a7be0acd52c9ad6e40939436477eba652faf35650e417beae00b704b","last_reissued_at":"2026-07-05T07:53:48.939286Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:48.939286Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can LLMs Follow Simple Rules?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Basel Alomair, Dan Hendrycks, David Karamardian, David Wagner, Lulwa Aljeraisy, Norman Mu, Sarah Chen, Sizhe Chen, Zifan Wang","submitted_at":"2023-11-06T08:50:29Z","abstract_excerpt":"As Large Language Models (LLMs) are deployed with increasing real-world responsibilities, it is important to be able to specify and constrain the behavior of these systems in a reliable manner. Model developers may wish to set explicit rules for the model, such as \"do not generate abusive content\", but these may be circumvented by jailbreaking techniques. Existing evaluations of adversarial attacks and defenses on LLMs generally require either expensive manual review or unreliable heuristic checks. To address this issue, we propose Rule-following Language Evaluation Scenarios (RuLES), a progra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.04235","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.04235/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.04235","created_at":"2026-07-05T07:53:48.939356+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.04235v3","created_at":"2026-07-05T07:53:48.939356+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.04235","created_at":"2026-07-05T07:53:48.939356+00:00"},{"alias_kind":"pith_short_12","alias_value":"5RC57MVHXYFM","created_at":"2026-07-05T07:53:48.939356+00:00"},{"alias_kind":"pith_short_16","alias_value":"5RC57MVHXYFM2UWJ","created_at":"2026-07-05T07:53:48.939356+00:00"},{"alias_kind":"pith_short_8","alias_value":"5RC57MVH","created_at":"2026-07-05T07:53:48.939356+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15874","citing_title":"LLM-as-Code: Agentic Programming for Agent Harness","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07539","citing_title":"Prompt Governance? On Governing Technologies Governed by Natural Language","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30783","citing_title":"Security--Fidelity Tradeoffs: The Hidden Cost of Prompt Injection Defense","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17510","citing_title":"Scale-Dependent Collective Adaptation in Self-Amending LLM Societies: A Cross-Family Study of Emergent Governance","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18333","citing_title":"Position: LLM Watermarking Should Align Stakeholders' Incentives for Practical Adoption","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2406.13352","citing_title":"AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09189","citing_title":"Do LLMs Follow Their Own Rules? A Reflexive Audit of Self-Stated Safety Policies","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ","json":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ.json","graph_json":"https://pith.science/api/pith-number/5RC57MVHXYFM2UWJVVXEBE4UGZ/graph.json","events_json":"https://pith.science/api/pith-number/5RC57MVHXYFM2UWJVVXEBE4UGZ/events.json","paper":"https://pith.science/paper/5RC57MVH"},"agent_actions":{"view_html":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ","download_json":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ.json","view_paper":"https://pith.science/paper/5RC57MVH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.04235&json=true","fetch_graph":"https://pith.science/api/pith-number/5RC57MVHXYFM2UWJVVXEBE4UGZ/graph.json","fetch_events":"https://pith.science/api/pith-number/5RC57MVHXYFM2UWJVVXEBE4UGZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ/action/storage_attestation","attest_author":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ/action/author_attestation","sign_citation":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ/action/citation_signature","submit_replication":"https://pith.science/pith/5RC57MVHXYFM2UWJVVXEBE4UGZ/action/replication_record"}},"created_at":"2026-07-05T07:53:48.939356+00:00","updated_at":"2026-07-05T07:53:48.939356+00:00"}