{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3J7SZEDFLDVDWUSO6VYGCG2LYI","short_pith_number":"pith:3J7SZEDF","schema_version":"1.0","canonical_sha256":"da7f2c906558ea3b524ef570611b4bc20691eddc24b03b0a828d3f174edc3be5","source":{"kind":"arxiv","id":"2501.06638","version":1},"attestation_state":"computed","paper":{"title":"Scaling Down Semantic Leakage: Investigating Associative Bias in Smaller Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Veronika Smilga","submitted_at":"2025-01-11T21:03:22Z","abstract_excerpt":"Semantic leakage is a phenomenon recently introduced by Gonen et al. (2024). It refers to a situation in which associations learnt from the training data emerge in language model generations in an unexpected and sometimes undesired way. Prior work has focused on leakage in large language models (7B+ parameters). In this study, I use Qwen2.5 model family to explore whether smaller models, ranging from 500M to 7B parameters, demonstrate less semantic leakage due to their limited capacity for capturing complex associations. Building on the previous dataset from Gonen et al. (2024), I introduce a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.06638","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-11T21:03:22Z","cross_cats_sorted":[],"title_canon_sha256":"8f2d8ba2c309f8bbf227b18a6caa403552fbbeb164668f8e08daaba0c8cd4e91","abstract_canon_sha256":"ea76cd5387087a1763abfb99bb0b654e834db41f65eb7ec9a05750b9e1bb9c41"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:07.631403Z","signature_b64":"DN5ajRQyOB3iFnMLx8q5wqfqpZqkj1x1+HNJMoO/A+BiH9G6cxfZgohQ8GlfN03qU4bRIpzVT/atDF30hKYMCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da7f2c906558ea3b524ef570611b4bc20691eddc24b03b0a828d3f174edc3be5","last_reissued_at":"2026-07-05T10:00:07.630966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:07.630966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Down Semantic Leakage: Investigating Associative Bias in Smaller Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Veronika Smilga","submitted_at":"2025-01-11T21:03:22Z","abstract_excerpt":"Semantic leakage is a phenomenon recently introduced by Gonen et al. (2024). It refers to a situation in which associations learnt from the training data emerge in language model generations in an unexpected and sometimes undesired way. Prior work has focused on leakage in large language models (7B+ parameters). In this study, I use Qwen2.5 model family to explore whether smaller models, ranging from 500M to 7B parameters, demonstrate less semantic leakage due to their limited capacity for capturing complex associations. Building on the previous dataset from Gonen et al. (2024), I introduce a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.06638","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.06638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.06638","created_at":"2026-07-05T10:00:07.631022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.06638v1","created_at":"2026-07-05T10:00:07.631022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.06638","created_at":"2026-07-05T10:00:07.631022+00:00"},{"alias_kind":"pith_short_12","alias_value":"3J7SZEDFLDVD","created_at":"2026-07-05T10:00:07.631022+00:00"},{"alias_kind":"pith_short_16","alias_value":"3J7SZEDFLDVDWUSO","created_at":"2026-07-05T10:00:07.631022+00:00"},{"alias_kind":"pith_short_8","alias_value":"3J7SZEDF","created_at":"2026-07-05T10:00:07.631022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10794","citing_title":"Can You Keep a Secret? Involuntary Information Leakage in Language Model Writing","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI","json":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI.json","graph_json":"https://pith.science/api/pith-number/3J7SZEDFLDVDWUSO6VYGCG2LYI/graph.json","events_json":"https://pith.science/api/pith-number/3J7SZEDFLDVDWUSO6VYGCG2LYI/events.json","paper":"https://pith.science/paper/3J7SZEDF"},"agent_actions":{"view_html":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI","download_json":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI.json","view_paper":"https://pith.science/paper/3J7SZEDF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.06638&json=true","fetch_graph":"https://pith.science/api/pith-number/3J7SZEDFLDVDWUSO6VYGCG2LYI/graph.json","fetch_events":"https://pith.science/api/pith-number/3J7SZEDFLDVDWUSO6VYGCG2LYI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI/action/storage_attestation","attest_author":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI/action/author_attestation","sign_citation":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI/action/citation_signature","submit_replication":"https://pith.science/pith/3J7SZEDFLDVDWUSO6VYGCG2LYI/action/replication_record"}},"created_at":"2026-07-05T10:00:07.631022+00:00","updated_at":"2026-07-05T10:00:07.631022+00:00"}