{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AIMOYO2U3UMSNETFNUF335OIM3","short_pith_number":"pith:AIMOYO2U","schema_version":"1.0","canonical_sha256":"0218ec3b54dd192692656d0bbdf5c866d2bdda74e4fc8dec451ef55ec9227f7a","source":{"kind":"arxiv","id":"2505.04606","version":1},"attestation_state":"computed","paper":{"title":"OmniGIRL: A Multilingual and Multimodal Benchmark for GitHub Issue Resolution","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Hongyu Zhang, Jiachi Chen, Lianghong Guo, Mingzhi Mao, Runhan Jiang, Wei Tao, Xilin Liu, Yanlin Wang, Yuchi Ma, Zibin Zheng","submitted_at":"2025-05-07T17:51:10Z","abstract_excerpt":"The GitHub issue resolution task aims to resolve issues reported in repositories automatically. With advances in large language models (LLMs), this task has gained increasing attention, and several benchmarks are proposed to evaluate the issue resolution ability of LLMs. However, existing benchmarks have three main limitations. First, current benchmarks focus on a single programming language, limiting the evaluation of issues from repositories across different languages. Second, they usually cover a narrow range of domains, which may fail to represent the diversity of real-world issues. Third,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.04606","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SE","submitted_at":"2025-05-07T17:51:10Z","cross_cats_sorted":[],"title_canon_sha256":"88ae861ff5a4046c638693fba942ba251c3d3f17485fdfacd6fd290f643f2948","abstract_canon_sha256":"91d6619fe52d54f85a9b7b78efc4b3ee9efeacd6fb105493b984c0e5cd20cb33"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:59:54.494228Z","signature_b64":"bZ51cCbh/tRKmjkeL1N5Spv99NnEOEs6IE5FOhuQ0LaNJaVaHqwHNtiGu6tRwKsfsTqy3EKB1OEoD1WtYnj2Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0218ec3b54dd192692656d0bbdf5c866d2bdda74e4fc8dec451ef55ec9227f7a","last_reissued_at":"2026-07-05T10:59:54.493686Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:59:54.493686Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OmniGIRL: A Multilingual and Multimodal Benchmark for GitHub Issue Resolution","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Hongyu Zhang, Jiachi Chen, Lianghong Guo, Mingzhi Mao, Runhan Jiang, Wei Tao, Xilin Liu, Yanlin Wang, Yuchi Ma, Zibin Zheng","submitted_at":"2025-05-07T17:51:10Z","abstract_excerpt":"The GitHub issue resolution task aims to resolve issues reported in repositories automatically. With advances in large language models (LLMs), this task has gained increasing attention, and several benchmarks are proposed to evaluate the issue resolution ability of LLMs. However, existing benchmarks have three main limitations. First, current benchmarks focus on a single programming language, limiting the evaluation of issues from repositories across different languages. Second, they usually cover a narrow range of domains, which may fail to represent the diversity of real-world issues. Third,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.04606","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.04606/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.04606","created_at":"2026-07-05T10:59:54.493746+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.04606v1","created_at":"2026-07-05T10:59:54.493746+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.04606","created_at":"2026-07-05T10:59:54.493746+00:00"},{"alias_kind":"pith_short_12","alias_value":"AIMOYO2U3UMS","created_at":"2026-07-05T10:59:54.493746+00:00"},{"alias_kind":"pith_short_16","alias_value":"AIMOYO2U3UMSNETF","created_at":"2026-07-05T10:59:54.493746+00:00"},{"alias_kind":"pith_short_8","alias_value":"AIMOYO2U","created_at":"2026-07-05T10:59:54.493746+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3","json":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3.json","graph_json":"https://pith.science/api/pith-number/AIMOYO2U3UMSNETFNUF335OIM3/graph.json","events_json":"https://pith.science/api/pith-number/AIMOYO2U3UMSNETFNUF335OIM3/events.json","paper":"https://pith.science/paper/AIMOYO2U"},"agent_actions":{"view_html":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3","download_json":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3.json","view_paper":"https://pith.science/paper/AIMOYO2U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.04606&json=true","fetch_graph":"https://pith.science/api/pith-number/AIMOYO2U3UMSNETFNUF335OIM3/graph.json","fetch_events":"https://pith.science/api/pith-number/AIMOYO2U3UMSNETFNUF335OIM3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3/action/storage_attestation","attest_author":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3/action/author_attestation","sign_citation":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3/action/citation_signature","submit_replication":"https://pith.science/pith/AIMOYO2U3UMSNETFNUF335OIM3/action/replication_record"}},"created_at":"2026-07-05T10:59:54.493746+00:00","updated_at":"2026-07-05T10:59:54.493746+00:00"}