{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HRARFLBGOQ4NYFH54XGCNMMCDM","short_pith_number":"pith:HRARFLBG","schema_version":"1.0","canonical_sha256":"3c4112ac267438dc14fde5cc26b1821b021455cfaf393fd6c13ff6e399841e25","source":{"kind":"arxiv","id":"2411.07668","version":3},"attestation_state":"computed","paper":{"title":"Towards Evaluation Guidelines for Empirical Studies involving LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Davide Falessi, Marvin Mu\\~noz Bar\\'on, Sebastian Baltes, Stefan Wagner","submitted_at":"2024-11-12T09:35:23Z","abstract_excerpt":"In the short period since the release of ChatGPT, large language models (LLMs) have changed the software engineering research landscape. While there are numerous opportunities to use LLMs for supporting research or software engineering tasks, solid science needs rigorous empirical evaluations. However, so far, there are no specific guidelines for conducting and assessing studies involving LLMs in software engineering research. Our focus is on empirical studies that either use LLMs as part of the research process or studies that evaluate existing or new tools that are based on LLMs. This paper "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.07668","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-11-12T09:35:23Z","cross_cats_sorted":[],"title_canon_sha256":"d56380cd05e49718d0c2d48c4d0bd66703f8c4cce2144d57a817a57d6b9c6415","abstract_canon_sha256":"aa45ef442897a6f8eff0cdc9e91e133567f1bd1a927bac651f9dc3c2fbfa532f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:11.895771Z","signature_b64":"op2vn/xiSiWYBTviHXmDOy+2kr58mcBPTCerfvVnAp0paMc4BUs54UmgawuaeHSKNRhSkJWS9ZoI6fq5OgXhAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c4112ac267438dc14fde5cc26b1821b021455cfaf393fd6c13ff6e399841e25","last_reissued_at":"2026-07-05T10:09:11.895299Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:11.895299Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Evaluation Guidelines for Empirical Studies involving LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Davide Falessi, Marvin Mu\\~noz Bar\\'on, Sebastian Baltes, Stefan Wagner","submitted_at":"2024-11-12T09:35:23Z","abstract_excerpt":"In the short period since the release of ChatGPT, large language models (LLMs) have changed the software engineering research landscape. While there are numerous opportunities to use LLMs for supporting research or software engineering tasks, solid science needs rigorous empirical evaluations. However, so far, there are no specific guidelines for conducting and assessing studies involving LLMs in software engineering research. Our focus is on empirical studies that either use LLMs as part of the research process or studies that evaluate existing or new tools that are based on LLMs. This paper "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.07668","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.07668/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.07668","created_at":"2026-07-05T10:09:11.895373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.07668v3","created_at":"2026-07-05T10:09:11.895373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.07668","created_at":"2026-07-05T10:09:11.895373+00:00"},{"alias_kind":"pith_short_12","alias_value":"HRARFLBGOQ4N","created_at":"2026-07-05T10:09:11.895373+00:00"},{"alias_kind":"pith_short_16","alias_value":"HRARFLBGOQ4NYFH5","created_at":"2026-07-05T10:09:11.895373+00:00"},{"alias_kind":"pith_short_8","alias_value":"HRARFLBG","created_at":"2026-07-05T10:09:11.895373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17637","citing_title":"Brick-DICL: Dynamic In-Context Learning for Automated Brick Schema Classification","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15139","citing_title":"Investigating Notable Metadata Practices in PyPI Libraries: An Empirical Study about Repository and Donation Platform URLs","ref_index":67,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM","json":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM.json","graph_json":"https://pith.science/api/pith-number/HRARFLBGOQ4NYFH54XGCNMMCDM/graph.json","events_json":"https://pith.science/api/pith-number/HRARFLBGOQ4NYFH54XGCNMMCDM/events.json","paper":"https://pith.science/paper/HRARFLBG"},"agent_actions":{"view_html":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM","download_json":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM.json","view_paper":"https://pith.science/paper/HRARFLBG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.07668&json=true","fetch_graph":"https://pith.science/api/pith-number/HRARFLBGOQ4NYFH54XGCNMMCDM/graph.json","fetch_events":"https://pith.science/api/pith-number/HRARFLBGOQ4NYFH54XGCNMMCDM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM/action/storage_attestation","attest_author":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM/action/author_attestation","sign_citation":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM/action/citation_signature","submit_replication":"https://pith.science/pith/HRARFLBGOQ4NYFH54XGCNMMCDM/action/replication_record"}},"created_at":"2026-07-05T10:09:11.895373+00:00","updated_at":"2026-07-05T10:09:11.895373+00:00"}