{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IACZMG77EMGECZZWAEJMS7V6FG","short_pith_number":"pith:IACZMG77","schema_version":"1.0","canonical_sha256":"4005961bff230c4167360112c97ebe299086f6e0d26dd9907c0f1781468bce3b","source":{"kind":"arxiv","id":"2507.18918","version":1},"attestation_state":"computed","paper":{"title":"Uncovering Cross-Linguistic Disparities in LLMs using Sparse Autoencoders","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jalil Huseynov, Richmond Sin Jing Xuan, Yang Zhang","submitted_at":"2025-07-25T03:22:50Z","abstract_excerpt":"Multilingual large language models (LLMs) exhibit strong cross-linguistic generalization, yet medium to low resource languages underperform on common benchmarks such as ARC-Challenge, MMLU, and HellaSwag. We analyze activation patterns in Gemma-2-2B across all 26 residual layers and 10 languages: Chinese (zh), Russian (ru), Spanish (es), Italian (it), medium to low resource languages including Indonesian (id), Catalan (ca), Marathi (mr), Malayalam (ml), and Hindi (hi), with English (en) as the reference. Using Sparse Autoencoders (SAEs), we reveal systematic disparities in activation patterns."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.18918","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-25T03:22:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"85fb402dd1aab7d937a03070d30f1a6c995cc60e96ea58f09a993903d69c8c12","abstract_canon_sha256":"94714e6f114ef1f655706876d1ce098cb8a779a577fee7b3d8593111a9d62a84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:16.197654Z","signature_b64":"p0Po2CMUGGfXIBwornuj0suB5OwiCO5/+koAUx61V5d03LrZhDBnNwIUkVzGx/6wyMKPA0XQK+y9cva6sIuBBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4005961bff230c4167360112c97ebe299086f6e0d26dd9907c0f1781468bce3b","last_reissued_at":"2026-07-05T11:43:16.197159Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:16.197159Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uncovering Cross-Linguistic Disparities in LLMs using Sparse Autoencoders","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jalil Huseynov, Richmond Sin Jing Xuan, Yang Zhang","submitted_at":"2025-07-25T03:22:50Z","abstract_excerpt":"Multilingual large language models (LLMs) exhibit strong cross-linguistic generalization, yet medium to low resource languages underperform on common benchmarks such as ARC-Challenge, MMLU, and HellaSwag. We analyze activation patterns in Gemma-2-2B across all 26 residual layers and 10 languages: Chinese (zh), Russian (ru), Spanish (es), Italian (it), medium to low resource languages including Indonesian (id), Catalan (ca), Marathi (mr), Malayalam (ml), and Hindi (hi), with English (en) as the reference. Using Sparse Autoencoders (SAEs), we reveal systematic disparities in activation patterns."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.18918","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.18918/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.18918","created_at":"2026-07-05T11:43:16.197218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.18918v1","created_at":"2026-07-05T11:43:16.197218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.18918","created_at":"2026-07-05T11:43:16.197218+00:00"},{"alias_kind":"pith_short_12","alias_value":"IACZMG77EMGE","created_at":"2026-07-05T11:43:16.197218+00:00"},{"alias_kind":"pith_short_16","alias_value":"IACZMG77EMGECZZW","created_at":"2026-07-05T11:43:16.197218+00:00"},{"alias_kind":"pith_short_8","alias_value":"IACZMG77","created_at":"2026-07-05T11:43:16.197218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG","json":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG.json","graph_json":"https://pith.science/api/pith-number/IACZMG77EMGECZZWAEJMS7V6FG/graph.json","events_json":"https://pith.science/api/pith-number/IACZMG77EMGECZZWAEJMS7V6FG/events.json","paper":"https://pith.science/paper/IACZMG77"},"agent_actions":{"view_html":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG","download_json":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG.json","view_paper":"https://pith.science/paper/IACZMG77","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.18918&json=true","fetch_graph":"https://pith.science/api/pith-number/IACZMG77EMGECZZWAEJMS7V6FG/graph.json","fetch_events":"https://pith.science/api/pith-number/IACZMG77EMGECZZWAEJMS7V6FG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG/action/storage_attestation","attest_author":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG/action/author_attestation","sign_citation":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG/action/citation_signature","submit_replication":"https://pith.science/pith/IACZMG77EMGECZZWAEJMS7V6FG/action/replication_record"}},"created_at":"2026-07-05T11:43:16.197218+00:00","updated_at":"2026-07-05T11:43:16.197218+00:00"}