{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HFDPO3EHVQ2WY2JZF7KLTA56LH","short_pith_number":"pith:HFDPO3EH","schema_version":"1.0","canonical_sha256":"3946f76c87ac356c69392fd4b983be59d98df38438c91ce908cd47128b4b8916","source":{"kind":"arxiv","id":"2507.00699","version":1},"attestation_state":"computed","paper":{"title":"A Hierarchical and Evolvable Benchmark for Fine-Grained Code Instruction Following with Multi-Turn Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Chong Wang, Guoliang Duan, Mingwei Liu, Xin Peng, Yanlin Wang, Zibin Zheng","submitted_at":"2025-07-01T11:51:40Z","abstract_excerpt":"Large language models (LLMs) have advanced significantly in code generation, yet their ability to follow complex programming instructions with layered and diverse constraints remains underexplored. Existing benchmarks often prioritize functional correctness, overlooking the nuanced requirements found in real-world development. We introduce MultiCodeIF, a comprehensive benchmark designed to evaluate instruction-following in code generation across multiple dimensions: constraint type, hierarchical levels, and iterative refinement. Built upon a structured taxonomy of 9 categories and 27 constrain"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.00699","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-07-01T11:51:40Z","cross_cats_sorted":[],"title_canon_sha256":"df838fcbb47786b67e48e67aec41ada15958737ed2487b3d6479e21f2635b797","abstract_canon_sha256":"6a6ff7adeecefb11e794e0bb0ade0fc4edbbf7ea12a95a1e40b8a84339c9eb7e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:13.171928Z","signature_b64":"EAJUyHwZqNdiC2am3txzylfneLgNfdtv07RnLTUi38Xl6BkECrn1q58eEIuHnX5ni0rFAi5XfHUIKlEVhwDnDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3946f76c87ac356c69392fd4b983be59d98df38438c91ce908cd47128b4b8916","last_reissued_at":"2026-07-05T11:30:13.171388Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:13.171388Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Hierarchical and Evolvable Benchmark for Fine-Grained Code Instruction Following with Multi-Turn Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Chong Wang, Guoliang Duan, Mingwei Liu, Xin Peng, Yanlin Wang, Zibin Zheng","submitted_at":"2025-07-01T11:51:40Z","abstract_excerpt":"Large language models (LLMs) have advanced significantly in code generation, yet their ability to follow complex programming instructions with layered and diverse constraints remains underexplored. Existing benchmarks often prioritize functional correctness, overlooking the nuanced requirements found in real-world development. We introduce MultiCodeIF, a comprehensive benchmark designed to evaluate instruction-following in code generation across multiple dimensions: constraint type, hierarchical levels, and iterative refinement. Built upon a structured taxonomy of 9 categories and 27 constrain"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00699","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.00699/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.00699","created_at":"2026-07-05T11:30:13.171464+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.00699v1","created_at":"2026-07-05T11:30:13.171464+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00699","created_at":"2026-07-05T11:30:13.171464+00:00"},{"alias_kind":"pith_short_12","alias_value":"HFDPO3EHVQ2W","created_at":"2026-07-05T11:30:13.171464+00:00"},{"alias_kind":"pith_short_16","alias_value":"HFDPO3EHVQ2WY2JZ","created_at":"2026-07-05T11:30:13.171464+00:00"},{"alias_kind":"pith_short_8","alias_value":"HFDPO3EH","created_at":"2026-07-05T11:30:13.171464+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25747","citing_title":"CodeChat-Eval: Evaluating Large Language Models in Multi-Turn Code Refinement Dialogues","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25747","citing_title":"CodeChat-Eval: Evaluating Large Language Models in Multi-Turn Code Refinement Dialogues","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07539","citing_title":"Prompt Governance? On Governing Technologies Governed by Natural Language","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30219","citing_title":"When Should Models Change Their Minds? Contextual Belief Management in Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH","json":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH.json","graph_json":"https://pith.science/api/pith-number/HFDPO3EHVQ2WY2JZF7KLTA56LH/graph.json","events_json":"https://pith.science/api/pith-number/HFDPO3EHVQ2WY2JZF7KLTA56LH/events.json","paper":"https://pith.science/paper/HFDPO3EH"},"agent_actions":{"view_html":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH","download_json":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH.json","view_paper":"https://pith.science/paper/HFDPO3EH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.00699&json=true","fetch_graph":"https://pith.science/api/pith-number/HFDPO3EHVQ2WY2JZF7KLTA56LH/graph.json","fetch_events":"https://pith.science/api/pith-number/HFDPO3EHVQ2WY2JZF7KLTA56LH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH/action/storage_attestation","attest_author":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH/action/author_attestation","sign_citation":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH/action/citation_signature","submit_replication":"https://pith.science/pith/HFDPO3EHVQ2WY2JZF7KLTA56LH/action/replication_record"}},"created_at":"2026-07-05T11:30:13.171464+00:00","updated_at":"2026-07-05T11:30:13.171464+00:00"}