{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HDXRMQ57CLCWZXMH724KMUVLCK","short_pith_number":"pith:HDXRMQ57","schema_version":"1.0","canonical_sha256":"38ef1643bf12c56cdd87feb8a652ab129548e46bc0e94f411afc6dc0efa8067f","source":{"kind":"arxiv","id":"2408.06518","version":3},"attestation_state":"computed","paper":{"title":"Does Liking Yellow Imply Driving a School Bus? Semantic Leakage in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alisa Liu, Hila Gonen, Luke Zettlemoyer, Noah A. Smith, Terra Blevins","submitted_at":"2024-08-12T22:30:55Z","abstract_excerpt":"Despite their wide adoption, the biases and unintended behaviors of language models remain poorly understood. In this paper, we identify and characterize a phenomenon never discussed before, which we call semantic leakage, where models leak irrelevant information from the prompt into the generation in unexpected ways. We propose an evaluation setting to detect semantic leakage both by humans and automatically, curate a diverse test suite for diagnosing this behavior, and measure significant semantic leakage in 13 flagship models. We also show that models exhibit semantic leakage in languages b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.06518","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-12T22:30:55Z","cross_cats_sorted":[],"title_canon_sha256":"d74eec9a2ca1fa9f8ceb4e5d5c334d77b62e53f87adf12f99098f0cb0de5d3e8","abstract_canon_sha256":"baab991a085a1e2decd9803771c88ca1d1e1f528c48dde570b22a68a69319d25"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:51.476309Z","signature_b64":"7YDbOgLHNSjNsn6XoRLOjR/4R15EIw2gKH8sYRuGl/XyawV+JwbN/R75SiURHgdXNwRGN2rFT3Aly4i3eavlDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38ef1643bf12c56cdd87feb8a652ab129548e46bc0e94f411afc6dc0efa8067f","last_reissued_at":"2026-07-05T11:03:51.475691Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:51.475691Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does Liking Yellow Imply Driving a School Bus? Semantic Leakage in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alisa Liu, Hila Gonen, Luke Zettlemoyer, Noah A. Smith, Terra Blevins","submitted_at":"2024-08-12T22:30:55Z","abstract_excerpt":"Despite their wide adoption, the biases and unintended behaviors of language models remain poorly understood. In this paper, we identify and characterize a phenomenon never discussed before, which we call semantic leakage, where models leak irrelevant information from the prompt into the generation in unexpected ways. We propose an evaluation setting to detect semantic leakage both by humans and automatically, curate a diverse test suite for diagnosing this behavior, and measure significant semantic leakage in 13 flagship models. We also show that models exhibit semantic leakage in languages b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.06518","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.06518/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.06518","created_at":"2026-07-05T11:03:51.475777+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.06518v3","created_at":"2026-07-05T11:03:51.475777+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.06518","created_at":"2026-07-05T11:03:51.475777+00:00"},{"alias_kind":"pith_short_12","alias_value":"HDXRMQ57CLCW","created_at":"2026-07-05T11:03:51.475777+00:00"},{"alias_kind":"pith_short_16","alias_value":"HDXRMQ57CLCWZXMH","created_at":"2026-07-05T11:03:51.475777+00:00"},{"alias_kind":"pith_short_8","alias_value":"HDXRMQ57","created_at":"2026-07-05T11:03:51.475777+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.05017","citing_title":"Verified Language Processing with Hybrid Explainability: A Technical Report","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK","json":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK.json","graph_json":"https://pith.science/api/pith-number/HDXRMQ57CLCWZXMH724KMUVLCK/graph.json","events_json":"https://pith.science/api/pith-number/HDXRMQ57CLCWZXMH724KMUVLCK/events.json","paper":"https://pith.science/paper/HDXRMQ57"},"agent_actions":{"view_html":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK","download_json":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK.json","view_paper":"https://pith.science/paper/HDXRMQ57","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.06518&json=true","fetch_graph":"https://pith.science/api/pith-number/HDXRMQ57CLCWZXMH724KMUVLCK/graph.json","fetch_events":"https://pith.science/api/pith-number/HDXRMQ57CLCWZXMH724KMUVLCK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK/action/storage_attestation","attest_author":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK/action/author_attestation","sign_citation":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK/action/citation_signature","submit_replication":"https://pith.science/pith/HDXRMQ57CLCWZXMH724KMUVLCK/action/replication_record"}},"created_at":"2026-07-05T11:03:51.475777+00:00","updated_at":"2026-07-05T11:03:51.475777+00:00"}