{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4HCJ4C7FZHUVFPGABOIBXNSUUI","short_pith_number":"pith:4HCJ4C7F","schema_version":"1.0","canonical_sha256":"e1c49e0be5c9e952bcc00b901bb654a22127261c0ec2ac70bcd8e2e118fbf3b6","source":{"kind":"arxiv","id":"2502.03708","version":2},"attestation_state":"computed","paper":{"title":"Toward universal steering and monitoring of AI models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.CL","authors_text":"Adityanarayanan Radhakrishnan, Daniel Beaglehole, Enric Boix-Adser\\`a, Mikhail Belkin","submitted_at":"2025-02-06T01:41:48Z","abstract_excerpt":"Modern AI models contain much of human knowledge, yet understanding of their internal representation of this knowledge remains elusive. Characterizing the structure and properties of this representation will lead to improvements in model capabilities and development of effective safeguards. Building on recent advances in feature learning, we develop an effective, scalable approach for extracting linear representations of general concepts in large-scale AI models (language models, vision-language models, and reasoning models). We show how these representations enable model steering, through whi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03708","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-06T01:41:48Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"1085f3fc262ba35ad579cd861232adb577308936a79b69dce73d685fae15a19d","abstract_canon_sha256":"b53732f4a1830086c9f0d1d59abb7de9747d634d445637c59ad4870209e1db4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:40.681532Z","signature_b64":"EB/bmy1M9S4Fo4Uhok1nNKVAxlfbVuWT6FrIuOWgk0lUw7CHfLmapLliVANIMpKl5daz6Xyd06Buw5o1cRklBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1c49e0be5c9e952bcc00b901bb654a22127261c0ec2ac70bcd8e2e118fbf3b6","last_reissued_at":"2026-07-05T11:11:40.681022Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:40.681022Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Toward universal steering and monitoring of AI models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.CL","authors_text":"Adityanarayanan Radhakrishnan, Daniel Beaglehole, Enric Boix-Adser\\`a, Mikhail Belkin","submitted_at":"2025-02-06T01:41:48Z","abstract_excerpt":"Modern AI models contain much of human knowledge, yet understanding of their internal representation of this knowledge remains elusive. Characterizing the structure and properties of this representation will lead to improvements in model capabilities and development of effective safeguards. Building on recent advances in feature learning, we develop an effective, scalable approach for extracting linear representations of general concepts in large-scale AI models (language models, vision-language models, and reasoning models). We show how these representations enable model steering, through whi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03708","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03708/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03708","created_at":"2026-07-05T11:11:40.681074+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03708v2","created_at":"2026-07-05T11:11:40.681074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03708","created_at":"2026-07-05T11:11:40.681074+00:00"},{"alias_kind":"pith_short_12","alias_value":"4HCJ4C7FZHUV","created_at":"2026-07-05T11:11:40.681074+00:00"},{"alias_kind":"pith_short_16","alias_value":"4HCJ4C7FZHUVFPGA","created_at":"2026-07-05T11:11:40.681074+00:00"},{"alias_kind":"pith_short_8","alias_value":"4HCJ4C7F","created_at":"2026-07-05T11:11:40.681074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00329","citing_title":"K-Inverse-RFM: A Modified RFM that Bridges the Gap to Neural Networks for Data-Corrupted Mathematical Tasks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28664","citing_title":"Activation Steering for Synthetic Data Generation: The Role of Diversity in Downstream Safety Detection","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10053","citing_title":"xRFM: Accurate, scalable, and interpretable feature learning models for tabular data","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19775","citing_title":"From Actions to Understanding: Conformal Interpretability of Temporal Concepts in LLM Agents","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI","json":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI.json","graph_json":"https://pith.science/api/pith-number/4HCJ4C7FZHUVFPGABOIBXNSUUI/graph.json","events_json":"https://pith.science/api/pith-number/4HCJ4C7FZHUVFPGABOIBXNSUUI/events.json","paper":"https://pith.science/paper/4HCJ4C7F"},"agent_actions":{"view_html":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI","download_json":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI.json","view_paper":"https://pith.science/paper/4HCJ4C7F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03708&json=true","fetch_graph":"https://pith.science/api/pith-number/4HCJ4C7FZHUVFPGABOIBXNSUUI/graph.json","fetch_events":"https://pith.science/api/pith-number/4HCJ4C7FZHUVFPGABOIBXNSUUI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI/action/storage_attestation","attest_author":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI/action/author_attestation","sign_citation":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI/action/citation_signature","submit_replication":"https://pith.science/pith/4HCJ4C7FZHUVFPGABOIBXNSUUI/action/replication_record"}},"created_at":"2026-07-05T11:11:40.681074+00:00","updated_at":"2026-07-05T11:11:40.681074+00:00"}