{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JQRVUTKUPHX3OZFXXCKWLF35BT","short_pith_number":"pith:JQRVUTKU","schema_version":"1.0","canonical_sha256":"4c235a4d5479efb764b7b89565977d0cc15484eea49617fd8b6bf1086d1b1ac1","source":{"kind":"arxiv","id":"2310.06762","version":1},"attestation_state":"computed","paper":{"title":"TRACE: A Comprehensive Benchmark for Continual Learning in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Qi Zhang, Rui Zheng, Senjie Jin, Songyang Gao, Tao Gui, Tianze Chen, Xianjun Yang, Xiao Wang, Xuanjing Huang, Yicheng Zou, Yuansen Zhang, Zhiheng Xi","submitted_at":"2023-10-10T16:38:49Z","abstract_excerpt":"Aligned large language models (LLMs) demonstrate exceptional capabilities in task-solving, following instructions, and ensuring safety. However, the continual learning aspect of these aligned LLMs has been largely overlooked. Existing continual learning benchmarks lack sufficient challenge for leading aligned LLMs, owing to both their simplicity and the models' potential exposure during instruction tuning. In this paper, we introduce TRACE, a novel benchmark designed to evaluate continual learning in LLMs. TRACE consists of 8 distinct datasets spanning challenging tasks including domain-specif"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06762","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-10T16:38:49Z","cross_cats_sorted":[],"title_canon_sha256":"da50520fb3dc357186f680948296d91a6743abc2bed286c71fbb8ac9643e1eec","abstract_canon_sha256":"dadb89d67703184b075cec61fad1f027ce2168dabe959c451809a1c60df77126"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:23.569772Z","signature_b64":"UljuCYNkEFPGu1ry6K8MHOXSN/yJQPiUmDAE4701llXAMRQYLUWrbbm0en9HRdaoLjFRgX+NAlGW79NPDPs/Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c235a4d5479efb764b7b89565977d0cc15484eea49617fd8b6bf1086d1b1ac1","last_reissued_at":"2026-07-05T06:59:23.569299Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:23.569299Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TRACE: A Comprehensive Benchmark for Continual Learning in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Qi Zhang, Rui Zheng, Senjie Jin, Songyang Gao, Tao Gui, Tianze Chen, Xianjun Yang, Xiao Wang, Xuanjing Huang, Yicheng Zou, Yuansen Zhang, Zhiheng Xi","submitted_at":"2023-10-10T16:38:49Z","abstract_excerpt":"Aligned large language models (LLMs) demonstrate exceptional capabilities in task-solving, following instructions, and ensuring safety. However, the continual learning aspect of these aligned LLMs has been largely overlooked. Existing continual learning benchmarks lack sufficient challenge for leading aligned LLMs, owing to both their simplicity and the models' potential exposure during instruction tuning. In this paper, we introduce TRACE, a novel benchmark designed to evaluate continual learning in LLMs. TRACE consists of 8 distinct datasets spanning challenging tasks including domain-specif"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06762","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06762/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06762","created_at":"2026-07-05T06:59:23.569381+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06762v1","created_at":"2026-07-05T06:59:23.569381+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06762","created_at":"2026-07-05T06:59:23.569381+00:00"},{"alias_kind":"pith_short_12","alias_value":"JQRVUTKUPHX3","created_at":"2026-07-05T06:59:23.569381+00:00"},{"alias_kind":"pith_short_16","alias_value":"JQRVUTKUPHX3OZFX","created_at":"2026-07-05T06:59:23.569381+00:00"},{"alias_kind":"pith_short_8","alias_value":"JQRVUTKU","created_at":"2026-07-05T06:59:23.569381+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21307","citing_title":"Task-Differentiated Atomic Skill Expansion and Routing for Continual Learning Across Highly Heterogeneous Tasks","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18543","citing_title":"CEO-Bench: Can Agents Play the Long Game?","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24901","citing_title":"LLM Evolution as an Industry-Scale Ecosystem: A Lifecycle Perspective on Continual Learning","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09255","citing_title":"RPO-PDT: Demonstrating Role-Play-Based Knowledge Adaptation for Student Support Dialogue (Demonstration System)","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06494","citing_title":"TailLoR: Protecting Principal Components in Parameter-Efficient Continual Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06698","citing_title":"RECAP: Regression Evaluation for Continual Adaptation of Prompts","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20638","citing_title":"RIZZ: Routing Interactions to Near Zero-Interference Zones for Continual Adaptation of Black-Box Agents","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20251","citing_title":"ProcCtrlBench: Evaluating Process-Level Defects and Control Preservation in LLM Coding Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31025","citing_title":"TRACE: Discovering Task-Specific Parameter via Adaptation-Aware Probing for Continual Fine-Tuning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00382","citing_title":"CRMA: A Spectrally-Bounded Backbone for Modular Continual Fine-Tuning of LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00400","citing_title":"Dynamic Proxy-Mixing: Transferring Replay Controllers from Small to Large Models for Continual Instruction Tuning","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15053","citing_title":"TFGN: Task-Free, Replay-Free Continual Pre-Training Without Catastrophic Forgetting at LLM Scale","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15384","citing_title":"Is One Score Enough? Rethinking the Evaluation of Sequentially Evolving LLM Memory","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08949","citing_title":"Muon-OGD: Muon-based Spectral Orthogonal Gradient Projection for LLM Continual Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.17183","citing_title":"LifeAlign: Lifelong Alignment for Large Language Models with Memory-Augmented Focalized Preference Optimization","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08949","citing_title":"Muon-OGD: Muon-based Spectral Orthogonal Gradient Projection for LLM Continual Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09734","citing_title":"Trajectory Supervision for Continual Tool-Use Learning in LLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05732","citing_title":"CRAFT: Forgetting-Aware Intervention-Based Adaptation for Continual Learning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05732","citing_title":"CRAFT: Forgetting-Aware Intervention-Based Adaptation for Continual Learning","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT","json":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT.json","graph_json":"https://pith.science/api/pith-number/JQRVUTKUPHX3OZFXXCKWLF35BT/graph.json","events_json":"https://pith.science/api/pith-number/JQRVUTKUPHX3OZFXXCKWLF35BT/events.json","paper":"https://pith.science/paper/JQRVUTKU"},"agent_actions":{"view_html":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT","download_json":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT.json","view_paper":"https://pith.science/paper/JQRVUTKU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06762&json=true","fetch_graph":"https://pith.science/api/pith-number/JQRVUTKUPHX3OZFXXCKWLF35BT/graph.json","fetch_events":"https://pith.science/api/pith-number/JQRVUTKUPHX3OZFXXCKWLF35BT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT/action/storage_attestation","attest_author":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT/action/author_attestation","sign_citation":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT/action/citation_signature","submit_replication":"https://pith.science/pith/JQRVUTKUPHX3OZFXXCKWLF35BT/action/replication_record"}},"created_at":"2026-07-05T06:59:23.569381+00:00","updated_at":"2026-07-05T06:59:23.569381+00:00"}