{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JUCDMNB2R2TD57CVAFMGYKA5DY","short_pith_number":"pith:JUCDMNB2","schema_version":"1.0","canonical_sha256":"4d0436343a8ea63efc5501586c281d1e1c97215b93ffc384f708870a00f40a2c","source":{"kind":"arxiv","id":"2409.00630","version":1},"attestation_state":"computed","paper":{"title":"LLMs as Evaluators: A Novel Approach to Evaluate Bug Report Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Abhishek Kumar, Partha Pratim Chakrabarti, Partha Pratim Das, Sonia Haiduc","submitted_at":"2024-09-01T06:30:39Z","abstract_excerpt":"Summarizing software artifacts is an important task that has been thoroughly researched. For evaluating software summarization approaches, human judgment is still the most trusted evaluation. However, it is time-consuming and fatiguing for evaluators, making it challenging to scale and reproduce. Large Language Models (LLMs) have demonstrated remarkable capabilities in various software engineering tasks, motivating us to explore their potential as automatic evaluators for approaches that aim to summarize software artifacts. In this study, we investigate whether LLMs can evaluate bug report sum"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00630","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-09-01T06:30:39Z","cross_cats_sorted":[],"title_canon_sha256":"e4b1bb97a21b56b8d9ddecfc0f03efeeb58ad741ddf252c58ba9bed056de611c","abstract_canon_sha256":"37c8ce6dd91f05b0e00c3893d5288211affc681d62020e6e7364c2ebf874754b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:50.266439Z","signature_b64":"EHRfRVNx1Y+i3Lv7cnzCDzbjkNuf0O2Xer1nbo+8blAh8vIC5Wg/PHToZJbkviqrxO1mIeOgF3gaTIEEY7t8CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d0436343a8ea63efc5501586c281d1e1c97215b93ffc384f708870a00f40a2c","last_reissued_at":"2026-07-05T09:01:50.265952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:50.265952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs as Evaluators: A Novel Approach to Evaluate Bug Report Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Abhishek Kumar, Partha Pratim Chakrabarti, Partha Pratim Das, Sonia Haiduc","submitted_at":"2024-09-01T06:30:39Z","abstract_excerpt":"Summarizing software artifacts is an important task that has been thoroughly researched. For evaluating software summarization approaches, human judgment is still the most trusted evaluation. However, it is time-consuming and fatiguing for evaluators, making it challenging to scale and reproduce. Large Language Models (LLMs) have demonstrated remarkable capabilities in various software engineering tasks, motivating us to explore their potential as automatic evaluators for approaches that aim to summarize software artifacts. In this study, we investigate whether LLMs can evaluate bug report sum"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00630","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00630/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00630","created_at":"2026-07-05T09:01:50.266009+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00630v1","created_at":"2026-07-05T09:01:50.266009+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00630","created_at":"2026-07-05T09:01:50.266009+00:00"},{"alias_kind":"pith_short_12","alias_value":"JUCDMNB2R2TD","created_at":"2026-07-05T09:01:50.266009+00:00"},{"alias_kind":"pith_short_16","alias_value":"JUCDMNB2R2TD57CV","created_at":"2026-07-05T09:01:50.266009+00:00"},{"alias_kind":"pith_short_8","alias_value":"JUCDMNB2","created_at":"2026-07-05T09:01:50.266009+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":119,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY","json":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY.json","graph_json":"https://pith.science/api/pith-number/JUCDMNB2R2TD57CVAFMGYKA5DY/graph.json","events_json":"https://pith.science/api/pith-number/JUCDMNB2R2TD57CVAFMGYKA5DY/events.json","paper":"https://pith.science/paper/JUCDMNB2"},"agent_actions":{"view_html":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY","download_json":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY.json","view_paper":"https://pith.science/paper/JUCDMNB2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00630&json=true","fetch_graph":"https://pith.science/api/pith-number/JUCDMNB2R2TD57CVAFMGYKA5DY/graph.json","fetch_events":"https://pith.science/api/pith-number/JUCDMNB2R2TD57CVAFMGYKA5DY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY/action/storage_attestation","attest_author":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY/action/author_attestation","sign_citation":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY/action/citation_signature","submit_replication":"https://pith.science/pith/JUCDMNB2R2TD57CVAFMGYKA5DY/action/replication_record"}},"created_at":"2026-07-05T09:01:50.266009+00:00","updated_at":"2026-07-05T09:01:50.266009+00:00"}