{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GWV2IF4ATOVZA6X3XGWE56YGUQ","short_pith_number":"pith:GWV2IF4A","schema_version":"1.0","canonical_sha256":"35aba417809bab907afbb9ac4efb06a4279c182afcc597b2b62cea185dbe38f3","source":{"kind":"arxiv","id":"2304.12918","version":1},"attestation_state":"computed","paper":{"title":"N2G: A Scalable Approach for Quantifying Interpretable Neuron Representations in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex Foote, Esben Kran, Fazl Barez, Ionnis Konstas, Neel Nanda","submitted_at":"2023-04-22T19:06:13Z","abstract_excerpt":"Understanding the function of individual neurons within language models is essential for mechanistic interpretability research. We propose $\\textbf{Neuron to Graph (N2G)}$, a tool which takes a neuron and its dataset examples, and automatically distills the neuron's behaviour on those examples to an interpretable graph. This presents a less labour intensive approach to interpreting neurons than current manual methods, that will better scale these methods to Large Language Models (LLMs). We use truncation and saliency methods to only present the important tokens, and augment the dataset example"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.12918","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-04-22T19:06:13Z","cross_cats_sorted":[],"title_canon_sha256":"20c1a7d3e51d2a2beb984c8264feb02230566b0fbc3fe537c5f25b198f5fb2e5","abstract_canon_sha256":"d845ef74b45203df9c7c28a189b1325fdbe57dbb414d6799e90723725af1348a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:04:22.454766Z","signature_b64":"m/wzOCH3S7dgsnf21AlAfDdh/LLGeYGP/y++0QpccrMROz1TxLC2y5ZdsRsLoYQ3ioITkhlhm+lomNNIpxSoDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35aba417809bab907afbb9ac4efb06a4279c182afcc597b2b62cea185dbe38f3","last_reissued_at":"2026-07-05T06:04:22.454277Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:04:22.454277Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"N2G: A Scalable Approach for Quantifying Interpretable Neuron Representations in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex Foote, Esben Kran, Fazl Barez, Ionnis Konstas, Neel Nanda","submitted_at":"2023-04-22T19:06:13Z","abstract_excerpt":"Understanding the function of individual neurons within language models is essential for mechanistic interpretability research. We propose $\\textbf{Neuron to Graph (N2G)}$, a tool which takes a neuron and its dataset examples, and automatically distills the neuron's behaviour on those examples to an interpretable graph. This presents a less labour intensive approach to interpreting neurons than current manual methods, that will better scale these methods to Large Language Models (LLMs). We use truncation and saliency methods to only present the important tokens, and augment the dataset example"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.12918","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.12918/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.12918","created_at":"2026-07-05T06:04:22.454352+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.12918v1","created_at":"2026-07-05T06:04:22.454352+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.12918","created_at":"2026-07-05T06:04:22.454352+00:00"},{"alias_kind":"pith_short_12","alias_value":"GWV2IF4ATOVZ","created_at":"2026-07-05T06:04:22.454352+00:00"},{"alias_kind":"pith_short_16","alias_value":"GWV2IF4ATOVZA6X3","created_at":"2026-07-05T06:04:22.454352+00:00"},{"alias_kind":"pith_short_8","alias_value":"GWV2IF4A","created_at":"2026-07-05T06:04:22.454352+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.06809","citing_title":"Neurons Speak in Ranges: Breaking Free from Discrete Neuronal Attribution","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":147,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ","json":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ.json","graph_json":"https://pith.science/api/pith-number/GWV2IF4ATOVZA6X3XGWE56YGUQ/graph.json","events_json":"https://pith.science/api/pith-number/GWV2IF4ATOVZA6X3XGWE56YGUQ/events.json","paper":"https://pith.science/paper/GWV2IF4A"},"agent_actions":{"view_html":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ","download_json":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ.json","view_paper":"https://pith.science/paper/GWV2IF4A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.12918&json=true","fetch_graph":"https://pith.science/api/pith-number/GWV2IF4ATOVZA6X3XGWE56YGUQ/graph.json","fetch_events":"https://pith.science/api/pith-number/GWV2IF4ATOVZA6X3XGWE56YGUQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ/action/storage_attestation","attest_author":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ/action/author_attestation","sign_citation":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ/action/citation_signature","submit_replication":"https://pith.science/pith/GWV2IF4ATOVZA6X3XGWE56YGUQ/action/replication_record"}},"created_at":"2026-07-05T06:04:22.454352+00:00","updated_at":"2026-07-05T06:04:22.454352+00:00"}