{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B42S5GZWLGCRQTP7OMLG3F2P4T","short_pith_number":"pith:B42S5GZW","schema_version":"1.0","canonical_sha256":"0f352e9b365985184dff73166d974fe4f945324bd1ed859888297ccfa424190e","source":{"kind":"arxiv","id":"2409.15820","version":2},"attestation_state":"computed","paper":{"title":"Supervised Fine-Tuning Achieve Rapid Task Adaption Via Alternating Attention Head Activation Patterns","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bing Qin, Kai Xiong, Li Du, Ting Liu, Xiao Ding, Yang Zhao","submitted_at":"2024-09-24T07:34:50Z","abstract_excerpt":"LLMs' performance on complex tasks is still unsatisfactory. A key issue is that presently LLMs learn in a data-driven schema, while the instructions about these complex tasks are both scarce and hard to collect or construct. On the contrary, a prominent phenomenon is that LLMs can learn rather fast on simpler tasks with adequate prior knowledge captured during pretraining stage. Thus, if the prerequisite and mechanism of such rapid generalization could be elucidated, it could enhance the efficiency and effectiveness of the LLM's ability to learn complex tasks. Thus, in this paper, we employ a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.15820","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-24T07:34:50Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5711ae6140fdc27d05ca409158a42f3718e786fc263ece0ecab3a716345a78d9","abstract_canon_sha256":"9ea04f62368c12ee3f08040e7d14791a08c9b78786dbacc7db8676f256882cc8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:21.941524Z","signature_b64":"y6u75PFt84ZTkV57vhvLpYaRMTq/0SNweeq2rbGX+Lz8orYogHuwg2gq9ZDdLXWpVoriCPEbBR6W+ebUIzmPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f352e9b365985184dff73166d974fe4f945324bd1ed859888297ccfa424190e","last_reissued_at":"2026-07-05T09:22:21.941021Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:21.941021Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Supervised Fine-Tuning Achieve Rapid Task Adaption Via Alternating Attention Head Activation Patterns","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bing Qin, Kai Xiong, Li Du, Ting Liu, Xiao Ding, Yang Zhao","submitted_at":"2024-09-24T07:34:50Z","abstract_excerpt":"LLMs' performance on complex tasks is still unsatisfactory. A key issue is that presently LLMs learn in a data-driven schema, while the instructions about these complex tasks are both scarce and hard to collect or construct. On the contrary, a prominent phenomenon is that LLMs can learn rather fast on simpler tasks with adequate prior knowledge captured during pretraining stage. Thus, if the prerequisite and mechanism of such rapid generalization could be elucidated, it could enhance the efficiency and effectiveness of the LLM's ability to learn complex tasks. Thus, in this paper, we employ a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15820","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.15820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.15820","created_at":"2026-07-05T09:22:21.941079+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.15820v2","created_at":"2026-07-05T09:22:21.941079+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15820","created_at":"2026-07-05T09:22:21.941079+00:00"},{"alias_kind":"pith_short_12","alias_value":"B42S5GZWLGCR","created_at":"2026-07-05T09:22:21.941079+00:00"},{"alias_kind":"pith_short_16","alias_value":"B42S5GZWLGCRQTP7","created_at":"2026-07-05T09:22:21.941079+00:00"},{"alias_kind":"pith_short_8","alias_value":"B42S5GZW","created_at":"2026-07-05T09:22:21.941079+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06632","citing_title":"Crafting Reversible SFT Behaviors in Large Language Models","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T","json":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T.json","graph_json":"https://pith.science/api/pith-number/B42S5GZWLGCRQTP7OMLG3F2P4T/graph.json","events_json":"https://pith.science/api/pith-number/B42S5GZWLGCRQTP7OMLG3F2P4T/events.json","paper":"https://pith.science/paper/B42S5GZW"},"agent_actions":{"view_html":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T","download_json":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T.json","view_paper":"https://pith.science/paper/B42S5GZW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.15820&json=true","fetch_graph":"https://pith.science/api/pith-number/B42S5GZWLGCRQTP7OMLG3F2P4T/graph.json","fetch_events":"https://pith.science/api/pith-number/B42S5GZWLGCRQTP7OMLG3F2P4T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T/action/storage_attestation","attest_author":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T/action/author_attestation","sign_citation":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T/action/citation_signature","submit_replication":"https://pith.science/pith/B42S5GZWLGCRQTP7OMLG3F2P4T/action/replication_record"}},"created_at":"2026-07-05T09:22:21.941079+00:00","updated_at":"2026-07-05T09:22:21.941079+00:00"}