{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:EPU4QWTSG4OPKT2KTHTQHDDDKM","short_pith_number":"pith:EPU4QWTS","schema_version":"1.0","canonical_sha256":"23e9c85a72371cf54f4a99e7038c63530e767d7734f98c5229a62bc5b6524835","source":{"kind":"arxiv","id":"2608.08570","version":1},"attestation_state":"computed","paper":{"title":"FailForge: Distilling Procedural Competence from Persistent Failures into Code Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Aichen Cai, Canhui Wu, Dongyi Lv, Fushun E, Jiaqi Wang, Liang Huang, Nan Duan, Qiuyu Ding, Ya Zhang, Yuesong Zhang, Zhi Wang","submitted_at":"2026-08-09T08:22:57Z","abstract_excerpt":"Rejection sampling fine-tuning (RFT) is widely used to train code agents by generating trajectories on verifiable software engineering tasks, retaining those that pass the tests, and fine-tuning on the successful rollouts. However, even strong code agents repeatedly fail on a substantial fraction of such tasks, and standard RFT simply discards these failures. The discarded samples are precisely the hardest and most informative ones, drawn from verifiable instances that are costly to curate. Stronger base models may reduce the number of failures, but the remaining hard cases still define the fr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.08570","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-08-09T08:22:57Z","cross_cats_sorted":[],"title_canon_sha256":"e58c2d969bd168e6d1c82b7701724befc1654504035594decaced631cffaddd0","abstract_canon_sha256":"416cfe4bce8aa29ba6a1d16242c9abe45a599ba6c8d6796970da725872f14272"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T01:23:00.872745Z","signature_b64":"hyExGfa8KTMArVbxaMzkp/+wCMPmfsrRPEGw5deYAQmTC330SZNpLrhRfiKWjEgaAI6QkED3ylRyJbkkvR45AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23e9c85a72371cf54f4a99e7038c63530e767d7734f98c5229a62bc5b6524835","last_reissued_at":"2026-08-11T01:23:00.870374Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T01:23:00.870374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FailForge: Distilling Procedural Competence from Persistent Failures into Code Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Aichen Cai, Canhui Wu, Dongyi Lv, Fushun E, Jiaqi Wang, Liang Huang, Nan Duan, Qiuyu Ding, Ya Zhang, Yuesong Zhang, Zhi Wang","submitted_at":"2026-08-09T08:22:57Z","abstract_excerpt":"Rejection sampling fine-tuning (RFT) is widely used to train code agents by generating trajectories on verifiable software engineering tasks, retaining those that pass the tests, and fine-tuning on the successful rollouts. However, even strong code agents repeatedly fail on a substantial fraction of such tasks, and standard RFT simply discards these failures. The discarded samples are precisely the hardest and most informative ones, drawn from verifiable instances that are costly to curate. Stronger base models may reduce the number of failures, but the remaining hard cases still define the fr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.08570","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.08570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.08570","created_at":"2026-08-11T01:23:00.871149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.08570v1","created_at":"2026-08-11T01:23:00.871149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.08570","created_at":"2026-08-11T01:23:00.871149+00:00"},{"alias_kind":"pith_short_12","alias_value":"EPU4QWTSG4OP","created_at":"2026-08-11T01:23:00.871149+00:00"},{"alias_kind":"pith_short_16","alias_value":"EPU4QWTSG4OPKT2K","created_at":"2026-08-11T01:23:00.871149+00:00"},{"alias_kind":"pith_short_8","alias_value":"EPU4QWTS","created_at":"2026-08-11T01:23:00.871149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM","json":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM.json","graph_json":"https://pith.science/api/pith-number/EPU4QWTSG4OPKT2KTHTQHDDDKM/graph.json","events_json":"https://pith.science/api/pith-number/EPU4QWTSG4OPKT2KTHTQHDDDKM/events.json","paper":"https://pith.science/paper/EPU4QWTS"},"agent_actions":{"view_html":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM","download_json":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM.json","view_paper":"https://pith.science/paper/EPU4QWTS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.08570&json=true","fetch_graph":"https://pith.science/api/pith-number/EPU4QWTSG4OPKT2KTHTQHDDDKM/graph.json","fetch_events":"https://pith.science/api/pith-number/EPU4QWTSG4OPKT2KTHTQHDDDKM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM/action/storage_attestation","attest_author":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM/action/author_attestation","sign_citation":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM/action/citation_signature","submit_replication":"https://pith.science/pith/EPU4QWTSG4OPKT2KTHTQHDDDKM/action/replication_record"}},"created_at":"2026-08-11T01:23:00.871149+00:00","updated_at":"2026-08-11T01:23:00.871149+00:00"}