{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GV7UZRHYGSLSBELXZRO6I2CLX7","short_pith_number":"pith:GV7UZRHY","schema_version":"1.0","canonical_sha256":"357f4cc4f83497209177cc5de4684bbfe255303ae0c1a474ce23b13688b48390","source":{"kind":"arxiv","id":"2501.17148","version":3},"attestation_state":"computed","paper":{"title":"AxBench: Steering LLMs? Even Simple Baselines Outperform Sparse Autoencoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aryaman Arora, Atticus Geiger, Christopher D. Manning, Christopher Potts, Dan Jurafsky, Jing Huang, Zheng Wang, Zhengxuan Wu","submitted_at":"2025-01-28T18:51:24Z","abstract_excerpt":"Fine-grained steering of language model outputs is essential for safety and reliability. Prompting and finetuning are widely used to achieve these goals, but interpretability researchers have proposed a variety of representation-based techniques as well, including sparse autoencoders (SAEs), linear artificial tomography, supervised steering vectors, linear probes, and representation finetuning. At present, there is no benchmark for making direct comparisons between these proposals. Therefore, we introduce AxBench, a large-scale benchmark for steering and concept detection, and report experimen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.17148","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-28T18:51:24Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"c768c5e61731523f55ff727450b0c2a56bbbecc65f10d8976d8910555c707d90","abstract_canon_sha256":"a02a5d3ad41c49e3131ffadd79da242f2241a923989838a9c76912b5d74c6255"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:28.862350Z","signature_b64":"Ca4Bmnj1JNQO09ELXJ8dhMb4ssAWu9uO4oMDpSGqjj2XbllpyxMuYkm037rXr/zMTlmbxE3tbVU+43asOF9pCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"357f4cc4f83497209177cc5de4684bbfe255303ae0c1a474ce23b13688b48390","last_reissued_at":"2026-07-05T10:23:28.861853Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:28.861853Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AxBench: Steering LLMs? Even Simple Baselines Outperform Sparse Autoencoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aryaman Arora, Atticus Geiger, Christopher D. Manning, Christopher Potts, Dan Jurafsky, Jing Huang, Zheng Wang, Zhengxuan Wu","submitted_at":"2025-01-28T18:51:24Z","abstract_excerpt":"Fine-grained steering of language model outputs is essential for safety and reliability. Prompting and finetuning are widely used to achieve these goals, but interpretability researchers have proposed a variety of representation-based techniques as well, including sparse autoencoders (SAEs), linear artificial tomography, supervised steering vectors, linear probes, and representation finetuning. At present, there is no benchmark for making direct comparisons between these proposals. Therefore, we introduce AxBench, a large-scale benchmark for steering and concept detection, and report experimen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.17148","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.17148/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.17148","created_at":"2026-07-05T10:23:28.861910+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.17148v3","created_at":"2026-07-05T10:23:28.861910+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.17148","created_at":"2026-07-05T10:23:28.861910+00:00"},{"alias_kind":"pith_short_12","alias_value":"GV7UZRHYGSLS","created_at":"2026-07-05T10:23:28.861910+00:00"},{"alias_kind":"pith_short_16","alias_value":"GV7UZRHYGSLSBELX","created_at":"2026-07-05T10:23:28.861910+00:00"},{"alias_kind":"pith_short_8","alias_value":"GV7UZRHY","created_at":"2026-07-05T10:23:28.861910+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11599","citing_title":"When is Your LLM Steerable?","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12360","citing_title":"Anatomy of Post-Training: Using Interpretability to Characterize Data and Shape the Learning Signal","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05194","citing_title":"Temporal Preference Concepts and their Functions in a Large Language Model","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15054","citing_title":"Size Doesn't Matter: Cosine-Scored Sparse Autoencoders","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28664","citing_title":"Activation Steering for Synthetic Data Generation: The Role of Diversity in Downstream Safety Detection","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16362","citing_title":"When Is Rank-1 Steering Cheap? Geometry, Granularity, and Budgeted Search","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16362","citing_title":"When Is Rank-1 Steering Cheap? Geometry, Granularity, and Budgeted Search","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18229","citing_title":"Are Sparse Autoencoder Benchmarks Reliable?","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12412","citing_title":"Stories in Space: In-Context Learning Trajectories in Conceptual Belief Space","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10601","citing_title":"The Open-Box Fallacy: Why AI Deployment Needs a Calibrated Verification Regime","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05715","citing_title":"Decodable but Not Corrected by Fixed Residual-Stream Linear Steering: Evidence from Medical LLM Failure Regimes","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7","json":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7.json","graph_json":"https://pith.science/api/pith-number/GV7UZRHYGSLSBELXZRO6I2CLX7/graph.json","events_json":"https://pith.science/api/pith-number/GV7UZRHYGSLSBELXZRO6I2CLX7/events.json","paper":"https://pith.science/paper/GV7UZRHY"},"agent_actions":{"view_html":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7","download_json":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7.json","view_paper":"https://pith.science/paper/GV7UZRHY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.17148&json=true","fetch_graph":"https://pith.science/api/pith-number/GV7UZRHYGSLSBELXZRO6I2CLX7/graph.json","fetch_events":"https://pith.science/api/pith-number/GV7UZRHYGSLSBELXZRO6I2CLX7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7/action/storage_attestation","attest_author":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7/action/author_attestation","sign_citation":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7/action/citation_signature","submit_replication":"https://pith.science/pith/GV7UZRHYGSLSBELXZRO6I2CLX7/action/replication_record"}},"created_at":"2026-07-05T10:23:28.861910+00:00","updated_at":"2026-07-05T10:23:28.861910+00:00"}