{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5NWL7WUNSFZKPJSBZLGDLLCKQX","short_pith_number":"pith:5NWL7WUN","schema_version":"1.0","canonical_sha256":"eb6cbfda8d9172a7a641cacc35ac4a85f3562a739b4b68d39c8df240fd429de0","source":{"kind":"arxiv","id":"2507.15889","version":1},"attestation_state":"computed","paper":{"title":"Dr. Boot: Bootstrapping Program Synthesis Language Models to Perform Repairing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Noah van der Vleuten","submitted_at":"2025-07-20T02:10:46Z","abstract_excerpt":"Language models for program synthesis are usually trained and evaluated on programming competition datasets (MBPP, APPS). However, these datasets are limited in size and quality, while these language models are extremely data hungry. Additionally, the language models have a misaligned program synthesis process compared to humans. While humans iteratively develop code with the help of a compiler, most program synthesis models currently produce code in one go. To solve these issues, we introduce a bootstrapping algorithm for program synthesis, that supports teaching models how to repair. We show"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15889","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-07-20T02:10:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a5419dce1195d8fd7c16fe612604138be15ed226bb8751f421b4f52d9082462b","abstract_canon_sha256":"9500055b54ab01015257cf4f70981d19813e786ec163d22310372f8c902d0318"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:58.673503Z","signature_b64":"tfmqFu/f8ZPErgLEAQJ9ABtkMz+C8bProUEJwlx30G06fw53rfPqD8IpmeQwK9KW9Y5FBexhR2nu0m9en8doDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb6cbfda8d9172a7a641cacc35ac4a85f3562a739b4b68d39c8df240fd429de0","last_reissued_at":"2026-07-05T11:40:58.672976Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:58.672976Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dr. Boot: Bootstrapping Program Synthesis Language Models to Perform Repairing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Noah van der Vleuten","submitted_at":"2025-07-20T02:10:46Z","abstract_excerpt":"Language models for program synthesis are usually trained and evaluated on programming competition datasets (MBPP, APPS). However, these datasets are limited in size and quality, while these language models are extremely data hungry. Additionally, the language models have a misaligned program synthesis process compared to humans. While humans iteratively develop code with the help of a compiler, most program synthesis models currently produce code in one go. To solve these issues, we introduce a bootstrapping algorithm for program synthesis, that supports teaching models how to repair. We show"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15889","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15889/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15889","created_at":"2026-07-05T11:40:58.673036+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15889v1","created_at":"2026-07-05T11:40:58.673036+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15889","created_at":"2026-07-05T11:40:58.673036+00:00"},{"alias_kind":"pith_short_12","alias_value":"5NWL7WUNSFZK","created_at":"2026-07-05T11:40:58.673036+00:00"},{"alias_kind":"pith_short_16","alias_value":"5NWL7WUNSFZKPJSB","created_at":"2026-07-05T11:40:58.673036+00:00"},{"alias_kind":"pith_short_8","alias_value":"5NWL7WUN","created_at":"2026-07-05T11:40:58.673036+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX","json":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX.json","graph_json":"https://pith.science/api/pith-number/5NWL7WUNSFZKPJSBZLGDLLCKQX/graph.json","events_json":"https://pith.science/api/pith-number/5NWL7WUNSFZKPJSBZLGDLLCKQX/events.json","paper":"https://pith.science/paper/5NWL7WUN"},"agent_actions":{"view_html":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX","download_json":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX.json","view_paper":"https://pith.science/paper/5NWL7WUN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15889&json=true","fetch_graph":"https://pith.science/api/pith-number/5NWL7WUNSFZKPJSBZLGDLLCKQX/graph.json","fetch_events":"https://pith.science/api/pith-number/5NWL7WUNSFZKPJSBZLGDLLCKQX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX/action/storage_attestation","attest_author":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX/action/author_attestation","sign_citation":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX/action/citation_signature","submit_replication":"https://pith.science/pith/5NWL7WUNSFZKPJSBZLGDLLCKQX/action/replication_record"}},"created_at":"2026-07-05T11:40:58.673036+00:00","updated_at":"2026-07-05T11:40:58.673036+00:00"}