{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XA3422FORPZTQOSGPMM733F7ED","short_pith_number":"pith:XA3422FO","schema_version":"1.0","canonical_sha256":"b837cd68ae8bf3383a467b19fdecbf20e27ffa1e73151068099fc45e672f986e","source":{"kind":"arxiv","id":"2509.03934","version":1},"attestation_state":"computed","paper":{"title":"SelfAug: Mitigating Catastrophic Forgetting in Retrieval-Augmented Generation via Distribution Self-Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chengqiang Lu, Enhong Chen, Guiquan Liu, Hao Wang, Qimeng Wang, Rongyang Zhang, Xin Li, Xuyang Zhi, Yan Gao, Yao Hu, Yi Wu, Yuqing Huang","submitted_at":"2025-09-04T06:50:47Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have revolutionized natural language processing through their remarkable capabilities in understanding and executing diverse tasks. While supervised fine-tuning, particularly in Retrieval-Augmented Generation (RAG) scenarios, effectively enhances task-specific performance, it often leads to catastrophic forgetting, where models lose their previously acquired knowledge and general capabilities. Existing solutions either require access to general instruction data or face limitations in preserving the model's original distribution. To overcome t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.03934","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-09-04T06:50:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4657c05e81a6a710db004de572f65f39469fc38d49ac5d82f70114d49505743d","abstract_canon_sha256":"717a8b225010d8f1992def7c525241e0ba980cb7ccebe9becca41166b04c198e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:04:50.113800Z","signature_b64":"Hj0tfyb54cfctpDRtacpATLFHlv6xRRPrsR96ePeM3vZrkumbEBol1xmm9tx9Z4vSRV3TGT1AUlNnPWgGi6LCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b837cd68ae8bf3383a467b19fdecbf20e27ffa1e73151068099fc45e672f986e","last_reissued_at":"2026-07-05T12:04:50.113263Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:04:50.113263Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SelfAug: Mitigating Catastrophic Forgetting in Retrieval-Augmented Generation via Distribution Self-Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chengqiang Lu, Enhong Chen, Guiquan Liu, Hao Wang, Qimeng Wang, Rongyang Zhang, Xin Li, Xuyang Zhi, Yan Gao, Yao Hu, Yi Wu, Yuqing Huang","submitted_at":"2025-09-04T06:50:47Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have revolutionized natural language processing through their remarkable capabilities in understanding and executing diverse tasks. While supervised fine-tuning, particularly in Retrieval-Augmented Generation (RAG) scenarios, effectively enhances task-specific performance, it often leads to catastrophic forgetting, where models lose their previously acquired knowledge and general capabilities. Existing solutions either require access to general instruction data or face limitations in preserving the model's original distribution. To overcome t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.03934","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.03934/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.03934","created_at":"2026-07-05T12:04:50.113344+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.03934v1","created_at":"2026-07-05T12:04:50.113344+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.03934","created_at":"2026-07-05T12:04:50.113344+00:00"},{"alias_kind":"pith_short_12","alias_value":"XA3422FORPZT","created_at":"2026-07-05T12:04:50.113344+00:00"},{"alias_kind":"pith_short_16","alias_value":"XA3422FORPZTQOSG","created_at":"2026-07-05T12:04:50.113344+00:00"},{"alias_kind":"pith_short_8","alias_value":"XA3422FO","created_at":"2026-07-05T12:04:50.113344+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.14930","citing_title":"IE as Cache: Information Extraction Enhanced Agentic Reasoning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15621","citing_title":"Rethinking the Necessity of Adaptive Retrieval-Augmented Generation through the Lens of Adaptive Listwise Ranking","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED","json":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED.json","graph_json":"https://pith.science/api/pith-number/XA3422FORPZTQOSGPMM733F7ED/graph.json","events_json":"https://pith.science/api/pith-number/XA3422FORPZTQOSGPMM733F7ED/events.json","paper":"https://pith.science/paper/XA3422FO"},"agent_actions":{"view_html":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED","download_json":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED.json","view_paper":"https://pith.science/paper/XA3422FO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.03934&json=true","fetch_graph":"https://pith.science/api/pith-number/XA3422FORPZTQOSGPMM733F7ED/graph.json","fetch_events":"https://pith.science/api/pith-number/XA3422FORPZTQOSGPMM733F7ED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED/action/storage_attestation","attest_author":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED/action/author_attestation","sign_citation":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED/action/citation_signature","submit_replication":"https://pith.science/pith/XA3422FORPZTQOSGPMM733F7ED/action/replication_record"}},"created_at":"2026-07-05T12:04:50.113344+00:00","updated_at":"2026-07-05T12:04:50.113344+00:00"}