{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FTJRKBFRXX56GKYB4ZDDMSV65J","short_pith_number":"pith:FTJRKBFR","schema_version":"1.0","canonical_sha256":"2cd31504b1bdfbe32b01e646364abeea66683aedb5e82582ecc21ae25dbafe7b","source":{"kind":"arxiv","id":"2509.00629","version":1},"attestation_state":"computed","paper":{"title":"Can Multi-turn Self-refined Single Agent LMs with Retrieval Solve Hard Coding Problems?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Md Kishor Morol, Md Tanzib Hosain","submitted_at":"2025-08-30T23:02:12Z","abstract_excerpt":"Among the hardest tasks for humans are those found in competitive programming where problems require sophisticated algorithmic thinking, puzzle solving, and the creation of effective code. As a domain to assess language models (LMs), it has not received enough attention, though. This study presents the ICPC benchmark, which consists of 254 international collegiate programming contest (ICPC) tasks. Each problem includes official analysis, reference code, and sample, high-quality unit, and hidden tests. We are able to develop and evaluate a variety of LM inference techniques for competitive prog"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.00629","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-30T23:02:12Z","cross_cats_sorted":[],"title_canon_sha256":"053a3674dae7393b7660bdb044a2ac5c630348851dcd6a12ca780e142bf5671b","abstract_canon_sha256":"2900c981783447b7cb40a08fdde367263ecd7e8491a54cd489bd056adf49eba2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:19.840520Z","signature_b64":"Y3e/0tE9lWFPxySaLBrxemTJHNLz74R9totxB1jXaaY5mRmBqcwSDCSlzCR/5cqLVROxE+x+99qecdeYldZ1BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cd31504b1bdfbe32b01e646364abeea66683aedb5e82582ecc21ae25dbafe7b","last_reissued_at":"2026-07-05T12:02:19.840101Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:19.840101Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Multi-turn Self-refined Single Agent LMs with Retrieval Solve Hard Coding Problems?","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Md Kishor Morol, Md Tanzib Hosain","submitted_at":"2025-08-30T23:02:12Z","abstract_excerpt":"Among the hardest tasks for humans are those found in competitive programming where problems require sophisticated algorithmic thinking, puzzle solving, and the creation of effective code. As a domain to assess language models (LMs), it has not received enough attention, though. This study presents the ICPC benchmark, which consists of 254 international collegiate programming contest (ICPC) tasks. Each problem includes official analysis, reference code, and sample, high-quality unit, and hidden tests. We are able to develop and evaluate a variety of LM inference techniques for competitive prog"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00629","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00629/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.00629","created_at":"2026-07-05T12:02:19.840165+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.00629v1","created_at":"2026-07-05T12:02:19.840165+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00629","created_at":"2026-07-05T12:02:19.840165+00:00"},{"alias_kind":"pith_short_12","alias_value":"FTJRKBFRXX56","created_at":"2026-07-05T12:02:19.840165+00:00"},{"alias_kind":"pith_short_16","alias_value":"FTJRKBFRXX56GKYB","created_at":"2026-07-05T12:02:19.840165+00:00"},{"alias_kind":"pith_short_8","alias_value":"FTJRKBFR","created_at":"2026-07-05T12:02:19.840165+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J","json":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J.json","graph_json":"https://pith.science/api/pith-number/FTJRKBFRXX56GKYB4ZDDMSV65J/graph.json","events_json":"https://pith.science/api/pith-number/FTJRKBFRXX56GKYB4ZDDMSV65J/events.json","paper":"https://pith.science/paper/FTJRKBFR"},"agent_actions":{"view_html":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J","download_json":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J.json","view_paper":"https://pith.science/paper/FTJRKBFR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.00629&json=true","fetch_graph":"https://pith.science/api/pith-number/FTJRKBFRXX56GKYB4ZDDMSV65J/graph.json","fetch_events":"https://pith.science/api/pith-number/FTJRKBFRXX56GKYB4ZDDMSV65J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J/action/storage_attestation","attest_author":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J/action/author_attestation","sign_citation":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J/action/citation_signature","submit_replication":"https://pith.science/pith/FTJRKBFRXX56GKYB4ZDDMSV65J/action/replication_record"}},"created_at":"2026-07-05T12:02:19.840165+00:00","updated_at":"2026-07-05T12:02:19.840165+00:00"}