{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:64BTHF2SVYANJGI5ZU2AQK3FNJ","short_pith_number":"pith:64BTHF2S","schema_version":"1.0","canonical_sha256":"f703339752ae00d4991dcd34082b656a686b46537a25dcda9d64f087ecbf4b6b","source":{"kind":"arxiv","id":"2502.06111","version":2},"attestation_state":"computed","paper":{"title":"CSR-Bench: Benchmarking LLM Agents in Deployment of Computer Science Research Repositories","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Davor Golac, Luyang Kong, Runhui Wang, Wei Wang, Yijia Xiao","submitted_at":"2025-02-10T02:46:29Z","abstract_excerpt":"The increasing complexity of computer science research projects demands more effective tools for deploying code repositories. Large Language Models (LLMs), such as Anthropic Claude and Meta Llama, have demonstrated significant advancements across various fields of computer science research, including the automation of diverse software engineering tasks. To evaluate the effectiveness of LLMs in handling complex code development tasks of research projects, particularly for NLP/CV/AI/ML/DM topics, we introduce CSR-Bench, a benchmark for Computer Science Research projects. This benchmark assesses "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06111","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SE","submitted_at":"2025-02-10T02:46:29Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"80f8f74ebf3f4ce3207b33badd0c15efeff70caad5c87cca5ea3c752f333e855","abstract_canon_sha256":"fb4b9f3c93dd45eae381c75ea608d5cb43a7067d068db051fa80de44ae2c5f71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:08.813759Z","signature_b64":"wi1U+PcwWMmBW0BAOW482I5BjlUdSQH04ncqYDBUfAYizY4AE+lTgWvflvuP7SiSxzmk7pn9ENC+AI9KlVUhDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f703339752ae00d4991dcd34082b656a686b46537a25dcda9d64f087ecbf4b6b","last_reissued_at":"2026-07-05T10:13:08.813166Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:08.813166Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CSR-Bench: Benchmarking LLM Agents in Deployment of Computer Science Research Repositories","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Davor Golac, Luyang Kong, Runhui Wang, Wei Wang, Yijia Xiao","submitted_at":"2025-02-10T02:46:29Z","abstract_excerpt":"The increasing complexity of computer science research projects demands more effective tools for deploying code repositories. Large Language Models (LLMs), such as Anthropic Claude and Meta Llama, have demonstrated significant advancements across various fields of computer science research, including the automation of diverse software engineering tasks. To evaluate the effectiveness of LLMs in handling complex code development tasks of research projects, particularly for NLP/CV/AI/ML/DM topics, we introduce CSR-Bench, a benchmark for Computer Science Research projects. This benchmark assesses "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06111","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06111/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06111","created_at":"2026-07-05T10:13:08.813234+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06111v2","created_at":"2026-07-05T10:13:08.813234+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06111","created_at":"2026-07-05T10:13:08.813234+00:00"},{"alias_kind":"pith_short_12","alias_value":"64BTHF2SVYAN","created_at":"2026-07-05T10:13:08.813234+00:00"},{"alias_kind":"pith_short_16","alias_value":"64BTHF2SVYANJGI5","created_at":"2026-07-05T10:13:08.813234+00:00"},{"alias_kind":"pith_short_8","alias_value":"64BTHF2S","created_at":"2026-07-05T10:13:08.813234+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.13896","citing_title":"GenCellAgent: Generalizable, Training-Free Cellular Image Segmentation via Large Language Model Agents","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ","json":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ.json","graph_json":"https://pith.science/api/pith-number/64BTHF2SVYANJGI5ZU2AQK3FNJ/graph.json","events_json":"https://pith.science/api/pith-number/64BTHF2SVYANJGI5ZU2AQK3FNJ/events.json","paper":"https://pith.science/paper/64BTHF2S"},"agent_actions":{"view_html":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ","download_json":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ.json","view_paper":"https://pith.science/paper/64BTHF2S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06111&json=true","fetch_graph":"https://pith.science/api/pith-number/64BTHF2SVYANJGI5ZU2AQK3FNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/64BTHF2SVYANJGI5ZU2AQK3FNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ/action/storage_attestation","attest_author":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ/action/author_attestation","sign_citation":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ/action/citation_signature","submit_replication":"https://pith.science/pith/64BTHF2SVYANJGI5ZU2AQK3FNJ/action/replication_record"}},"created_at":"2026-07-05T10:13:08.813234+00:00","updated_at":"2026-07-05T10:13:08.813234+00:00"}