{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q3CCEY5ZXVZ7K55MMUBIVCCUXB","short_pith_number":"pith:Q3CCEY5Z","schema_version":"1.0","canonical_sha256":"86c42263b9bd73f577ac65028a8854b865c246aa1b26dbcc8108580c12ec7674","source":{"kind":"arxiv","id":"2502.11356","version":1},"attestation_state":"computed","paper":{"title":"SAIF: A Sparse Autoencoder Framework for Interpreting and Steering Instruction Following of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ali Payani, Fan Yang, Haiyan Zhao, Jing Ma, Mengnan Du, Yiran Qiao, Zirui He","submitted_at":"2025-02-17T02:11:17Z","abstract_excerpt":"The ability of large language models (LLMs) to follow instructions is crucial for their practical applications, yet the underlying mechanisms remain poorly understood. This paper presents a novel framework that leverages sparse autoencoders (SAE) to interpret how instruction following works in these models. We demonstrate how the features we identify can effectively steer model outputs to align with given instructions. Through analysis of SAE latent activations, we identify specific latents responsible for instruction following behavior. Our findings reveal that instruction following capabilit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11356","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-17T02:11:17Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"bbaf164023d2f8839b2aefe48fbb5a47833b7c730d684cfc8ae55d6760778e92","abstract_canon_sha256":"eac743c167fe4e336e11d4390a06e1c1a02a80b944f8dfcdddc6c37c9c1115ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:27.089962Z","signature_b64":"QlHOAvBU6cOIvpKve0FS9xtaPxlqtPhxgUrrzo1aWKgMdpngnaCh0Ge4KPGqqyDzpIH21dtcFjKwUVwwd+FbBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86c42263b9bd73f577ac65028a8854b865c246aa1b26dbcc8108580c12ec7674","last_reissued_at":"2026-07-05T10:15:27.089473Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:27.089473Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SAIF: A Sparse Autoencoder Framework for Interpreting and Steering Instruction Following of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ali Payani, Fan Yang, Haiyan Zhao, Jing Ma, Mengnan Du, Yiran Qiao, Zirui He","submitted_at":"2025-02-17T02:11:17Z","abstract_excerpt":"The ability of large language models (LLMs) to follow instructions is crucial for their practical applications, yet the underlying mechanisms remain poorly understood. This paper presents a novel framework that leverages sparse autoencoders (SAE) to interpret how instruction following works in these models. We demonstrate how the features we identify can effectively steer model outputs to align with given instructions. Through analysis of SAE latent activations, we identify specific latents responsible for instruction following behavior. Our findings reveal that instruction following capabilit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11356","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11356","created_at":"2026-07-05T10:15:27.089533+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11356v1","created_at":"2026-07-05T10:15:27.089533+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11356","created_at":"2026-07-05T10:15:27.089533+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q3CCEY5ZXVZ7","created_at":"2026-07-05T10:15:27.089533+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q3CCEY5ZXVZ7K55M","created_at":"2026-07-05T10:15:27.089533+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q3CCEY5Z","created_at":"2026-07-05T10:15:27.089533+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11887","citing_title":"Qwen-Scope: Turning Sparse Features into Development Tools for Large Language Models","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB","json":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB.json","graph_json":"https://pith.science/api/pith-number/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/graph.json","events_json":"https://pith.science/api/pith-number/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/events.json","paper":"https://pith.science/paper/Q3CCEY5Z"},"agent_actions":{"view_html":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB","download_json":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB.json","view_paper":"https://pith.science/paper/Q3CCEY5Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11356&json=true","fetch_graph":"https://pith.science/api/pith-number/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/graph.json","fetch_events":"https://pith.science/api/pith-number/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/action/storage_attestation","attest_author":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/action/author_attestation","sign_citation":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/action/citation_signature","submit_replication":"https://pith.science/pith/Q3CCEY5ZXVZ7K55MMUBIVCCUXB/action/replication_record"}},"created_at":"2026-07-05T10:15:27.089533+00:00","updated_at":"2026-07-05T10:15:27.089533+00:00"}