{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZTGYRLS3TRJ2XOGSEEFPHGZXDK","short_pith_number":"pith:ZTGYRLS3","schema_version":"1.0","canonical_sha256":"cccd88ae5b9c53abb8d2210af39b371a8fc8850f5c8647b2f3beb75fe98a7665","source":{"kind":"arxiv","id":"2405.17799","version":1},"attestation_state":"computed","paper":{"title":"Exploring Activation Patterns of Parameters in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Damai Dai, Yudong Wang, Zhifang Sui","submitted_at":"2024-05-28T03:49:54Z","abstract_excerpt":"Most work treats large language models as black boxes without in-depth understanding of their internal working mechanism. In order to explain the internal representations of LLMs, we propose a gradient-based metric to assess the activation level of model parameters. Based on this metric, we obtain three preliminary findings. (1) When the inputs are in the same domain, parameters in the shallow layers will be activated densely, which means a larger portion of parameters will have great impacts on the outputs. In contrast, parameters in the deep layers are activated sparsely. (2) When the inputs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17799","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-28T03:49:54Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"017dd10e00fa3e8d6b2d4d9967772c1989c4d545bfea08e6b80abdacc7ddabbc","abstract_canon_sha256":"7fc625ba8fa8e9289011674098619a0e7c1caac96cb8aa3f29b98aa970bd5dd1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:11.498099Z","signature_b64":"U7Rwlt8stMg7TDS30ajx61/IVtFF5tfu+DeRDY10dfuxubwRLnPQSSu4wZ5kWtlHN7uInem+QetgQIzEjIsVCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cccd88ae5b9c53abb8d2210af39b371a8fc8850f5c8647b2f3beb75fe98a7665","last_reissued_at":"2026-07-05T08:24:11.497665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:11.497665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Activation Patterns of Parameters in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Damai Dai, Yudong Wang, Zhifang Sui","submitted_at":"2024-05-28T03:49:54Z","abstract_excerpt":"Most work treats large language models as black boxes without in-depth understanding of their internal working mechanism. In order to explain the internal representations of LLMs, we propose a gradient-based metric to assess the activation level of model parameters. Based on this metric, we obtain three preliminary findings. (1) When the inputs are in the same domain, parameters in the shallow layers will be activated densely, which means a larger portion of parameters will have great impacts on the outputs. In contrast, parameters in the deep layers are activated sparsely. (2) When the inputs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17799","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17799","created_at":"2026-07-05T08:24:11.497740+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17799v1","created_at":"2026-07-05T08:24:11.497740+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17799","created_at":"2026-07-05T08:24:11.497740+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZTGYRLS3TRJ2","created_at":"2026-07-05T08:24:11.497740+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZTGYRLS3TRJ2XOGS","created_at":"2026-07-05T08:24:11.497740+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZTGYRLS3","created_at":"2026-07-05T08:24:11.497740+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23862","citing_title":"Graph Memory Transformer (GMT)","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK","json":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK.json","graph_json":"https://pith.science/api/pith-number/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/graph.json","events_json":"https://pith.science/api/pith-number/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/events.json","paper":"https://pith.science/paper/ZTGYRLS3"},"agent_actions":{"view_html":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK","download_json":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK.json","view_paper":"https://pith.science/paper/ZTGYRLS3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17799&json=true","fetch_graph":"https://pith.science/api/pith-number/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/graph.json","fetch_events":"https://pith.science/api/pith-number/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/action/storage_attestation","attest_author":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/action/author_attestation","sign_citation":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/action/citation_signature","submit_replication":"https://pith.science/pith/ZTGYRLS3TRJ2XOGSEEFPHGZXDK/action/replication_record"}},"created_at":"2026-07-05T08:24:11.497740+00:00","updated_at":"2026-07-05T08:24:11.497740+00:00"}