{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:VSTYNZ3GVS3574RCUFN3LLVV4R","short_pith_number":"pith:VSTYNZ3G","canonical_record":{"source":{"id":"2512.10092","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-10T21:26:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7cdc26b7ce018a6f600d23d193cbd2d78ee947bc68aac5cc70b072a8fce293b9","abstract_canon_sha256":"1aa94c1bd47c01cbb80ec85f8653350fbd6ee97769d669e7336dd8297c59009e"},"schema_version":"1.0"},"canonical_sha256":"aca786e766acb7dff222a15bb5aeb5e45f6a3fe3fad1606816bc484439895152","source":{"kind":"arxiv","id":"2512.10092","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2512.10092","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"arxiv_version","alias_value":"2512.10092v2","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.10092","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"pith_short_12","alias_value":"VSTYNZ3GVS35","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"pith_short_16","alias_value":"VSTYNZ3GVS3574RC","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"pith_short_8","alias_value":"VSTYNZ3G","created_at":"2026-07-24T00:23:06Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:VSTYNZ3GVS3574RCUFN3LLVV4R","target":"record","payload":{"canonical_record":{"source":{"id":"2512.10092","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-10T21:26:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7cdc26b7ce018a6f600d23d193cbd2d78ee947bc68aac5cc70b072a8fce293b9","abstract_canon_sha256":"1aa94c1bd47c01cbb80ec85f8653350fbd6ee97769d669e7336dd8297c59009e"},"schema_version":"1.0"},"canonical_sha256":"aca786e766acb7dff222a15bb5aeb5e45f6a3fe3fad1606816bc484439895152","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-24T00:23:06.493234Z","signature_b64":"5m2o2ERpOr9PEQKWejjgV52UmSzrp0dgpShwMBxWWEQRVlUhhrtCRw//tVmpcOGPe8TIFSCfCePRt/I675zGAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aca786e766acb7dff222a15bb5aeb5e45f6a3fe3fad1606816bc484439895152","last_reissued_at":"2026-07-24T00:23:06.492202Z","signature_status":"signed_v1","first_computed_at":"2026-07-24T00:23:06.492202Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2512.10092","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-24T00:23:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9BTfbR/o+nkcVLHYvYZbG/lMRcwSD4jTJDhUCCJ3IfbYUpTcbqglTzYkL2tA+HbMzMPl35po/uo8ita2tgUNAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T22:51:34.041276Z"},"content_sha256":"c1af6546b6569d951eae6842a886d3a8438ec543d154838ae918052d492bfe8d","schema_version":"1.0","event_id":"sha256:c1af6546b6569d951eae6842a886d3a8438ec543d154838ae918052d492bfe8d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:VSTYNZ3GVS3574RCUFN3LLVV4R","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Interpretable Embeddings with Sparse Autoencoders: A Data Analysis Toolkit","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Lewis Smith, Lisa Dunlap, Neel Nanda, Nick Jiang, Xiaoqing Sun","submitted_at":"2025-12-10T21:26:24Z","abstract_excerpt":"Analyzing large-scale text corpora is a core challenge in machine learning, crucial for tasks like identifying undesirable model behaviors or biases in training data. Current methods often rely on costly LLM-based techniques (e.g. annotating dataset differences) or dense embedding models (e.g. for clustering), which lack control over the properties of interest. We propose using sparse autoencoders (SAEs) to create SAE embeddings: representations whose dimensions map to interpretable concepts. Through four data analysis tasks, we show that SAE embeddings are more cost-effective and reliable tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.10092","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.10092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-24T00:23:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WSVrYKX+lE961Cf/WCce8E5vLrmRhz3hr049/RKrLGnh+NRCw/QqWyJBUM5BTCm/qY3gCGwtKmmk4xEcPmP7Bg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T22:51:34.041821Z"},"content_sha256":"3a91c2c5d879bb16e965d7e7070be7766cea37d4cf92a08e5c578c6ef26a5275","schema_version":"1.0","event_id":"sha256:3a91c2c5d879bb16e965d7e7070be7766cea37d4cf92a08e5c578c6ef26a5275"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/VSTYNZ3GVS3574RCUFN3LLVV4R/bundle.json","state_url":"https://pith.science/pith/VSTYNZ3GVS3574RCUFN3LLVV4R/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/VSTYNZ3GVS3574RCUFN3LLVV4R/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T22:51:34Z","links":{"resolver":"https://pith.science/pith/VSTYNZ3GVS3574RCUFN3LLVV4R","bundle":"https://pith.science/pith/VSTYNZ3GVS3574RCUFN3LLVV4R/bundle.json","state":"https://pith.science/pith/VSTYNZ3GVS3574RCUFN3LLVV4R/state.json","well_known_bundle":"https://pith.science/.well-known/pith/VSTYNZ3GVS3574RCUFN3LLVV4R/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:VSTYNZ3GVS3574RCUFN3LLVV4R","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1aa94c1bd47c01cbb80ec85f8653350fbd6ee97769d669e7336dd8297c59009e","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-10T21:26:24Z","title_canon_sha256":"7cdc26b7ce018a6f600d23d193cbd2d78ee947bc68aac5cc70b072a8fce293b9"},"schema_version":"1.0","source":{"id":"2512.10092","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2512.10092","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"arxiv_version","alias_value":"2512.10092v2","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.10092","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"pith_short_12","alias_value":"VSTYNZ3GVS35","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"pith_short_16","alias_value":"VSTYNZ3GVS3574RC","created_at":"2026-07-24T00:23:06Z"},{"alias_kind":"pith_short_8","alias_value":"VSTYNZ3G","created_at":"2026-07-24T00:23:06Z"}],"graph_snapshots":[{"event_id":"sha256:3a91c2c5d879bb16e965d7e7070be7766cea37d4cf92a08e5c578c6ef26a5275","target":"graph","created_at":"2026-07-24T00:23:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2512.10092/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Analyzing large-scale text corpora is a core challenge in machine learning, crucial for tasks like identifying undesirable model behaviors or biases in training data. Current methods often rely on costly LLM-based techniques (e.g. annotating dataset differences) or dense embedding models (e.g. for clustering), which lack control over the properties of interest. We propose using sparse autoencoders (SAEs) to create SAE embeddings: representations whose dimensions map to interpretable concepts. Through four data analysis tasks, we show that SAE embeddings are more cost-effective and reliable tha","authors_text":"Lewis Smith, Lisa Dunlap, Neel Nanda, Nick Jiang, Xiaoqing Sun","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-10T21:26:24Z","title":"Interpretable Embeddings with Sparse Autoencoders: A Data Analysis Toolkit"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.10092","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c1af6546b6569d951eae6842a886d3a8438ec543d154838ae918052d492bfe8d","target":"record","created_at":"2026-07-24T00:23:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1aa94c1bd47c01cbb80ec85f8653350fbd6ee97769d669e7336dd8297c59009e","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-12-10T21:26:24Z","title_canon_sha256":"7cdc26b7ce018a6f600d23d193cbd2d78ee947bc68aac5cc70b072a8fce293b9"},"schema_version":"1.0","source":{"id":"2512.10092","kind":"arxiv","version":2}},"canonical_sha256":"aca786e766acb7dff222a15bb5aeb5e45f6a3fe3fad1606816bc484439895152","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"aca786e766acb7dff222a15bb5aeb5e45f6a3fe3fad1606816bc484439895152","first_computed_at":"2026-07-24T00:23:06.492202Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-24T00:23:06.492202Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"5m2o2ERpOr9PEQKWejjgV52UmSzrp0dgpShwMBxWWEQRVlUhhrtCRw//tVmpcOGPe8TIFSCfCePRt/I675zGAQ==","signature_status":"signed_v1","signed_at":"2026-07-24T00:23:06.493234Z","signed_message":"canonical_sha256_bytes"},"source_id":"2512.10092","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c1af6546b6569d951eae6842a886d3a8438ec543d154838ae918052d492bfe8d","sha256:3a91c2c5d879bb16e965d7e7070be7766cea37d4cf92a08e5c578c6ef26a5275"],"state_sha256":"b3261eee303df40403e600141cac606cf152b75e1e7ad442764d480dab934ed5"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mtpyHtGWA+sndPd3lOCGCaKyjN+B8OzvkSJkn2phWPlNABNfF1argANt0/Xheq15X2Qz8dlSAp0ZfCsJ1SAGAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T22:51:34.047751Z","bundle_sha256":"7d1fae8c02ed86e73d7dc345c4b0bc0361d60e185d81772b621353dd8469c22a"}}