{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6TLMA5MESRVURSOB3CDBJJSCIX","short_pith_number":"pith:6TLMA5ME","schema_version":"1.0","canonical_sha256":"f4d6c07584946b48c9c1d88614a64245d6dc58fbfff1b2fe8c1cf8a0fe37a5e5","source":{"kind":"arxiv","id":"2405.14445","version":2},"attestation_state":"computed","paper":{"title":"Exploring the use of a Large Language Model for data extraction in systematic reviews: a rapid feasibility study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"AliReza Khanteymoori, Claudia Kapp, Dawn Craig, Fiona Campbell, James Thomas, Kaitlyn Hair, Lena Schmidt, Mark Engelbert, Sergio Graziosi","submitted_at":"2024-05-23T11:24:23Z","abstract_excerpt":"This paper describes a rapid feasibility study of using GPT-4, a large language model (LLM), to (semi)automate data extraction in systematic reviews. Despite the recent surge of interest in LLMs there is still a lack of understanding of how to design LLM-based automation tools and how to robustly evaluate their performance. During the 2023 Evidence Synthesis Hackathon we conducted two feasibility studies. Firstly, to automatically extract study characteristics from human clinical, animal, and social science domain studies. We used two studies from each category for prompt-development; and ten "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14445","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-23T11:24:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4a143b497ca69833c7885c7fc7676401917ace1db646b263ecef17e7a23d7afe","abstract_canon_sha256":"16fb21e83b7633d49533683f849eae062ff2bd3e2601bac72e3fb447bba990c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:40.182046Z","signature_b64":"Am9FbST/RyUl+4+Vo9F2SJ9q29FDZtYYILdzN/yt+4c+D19YtjXfzE3S41bhtCzuD9v1l46F8O0vkFY5ODurAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4d6c07584946b48c9c1d88614a64245d6dc58fbfff1b2fe8c1cf8a0fe37a5e5","last_reissued_at":"2026-07-05T10:13:40.181560Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:40.181560Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the use of a Large Language Model for data extraction in systematic reviews: a rapid feasibility study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"AliReza Khanteymoori, Claudia Kapp, Dawn Craig, Fiona Campbell, James Thomas, Kaitlyn Hair, Lena Schmidt, Mark Engelbert, Sergio Graziosi","submitted_at":"2024-05-23T11:24:23Z","abstract_excerpt":"This paper describes a rapid feasibility study of using GPT-4, a large language model (LLM), to (semi)automate data extraction in systematic reviews. Despite the recent surge of interest in LLMs there is still a lack of understanding of how to design LLM-based automation tools and how to robustly evaluate their performance. During the 2023 Evidence Synthesis Hackathon we conducted two feasibility studies. Firstly, to automatically extract study characteristics from human clinical, animal, and social science domain studies. We used two studies from each category for prompt-development; and ten "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14445","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14445/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14445","created_at":"2026-07-05T10:13:40.181611+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14445v2","created_at":"2026-07-05T10:13:40.181611+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14445","created_at":"2026-07-05T10:13:40.181611+00:00"},{"alias_kind":"pith_short_12","alias_value":"6TLMA5MESRVU","created_at":"2026-07-05T10:13:40.181611+00:00"},{"alias_kind":"pith_short_16","alias_value":"6TLMA5MESRVURSOB","created_at":"2026-07-05T10:13:40.181611+00:00"},{"alias_kind":"pith_short_8","alias_value":"6TLMA5ME","created_at":"2026-07-05T10:13:40.181611+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.11840","citing_title":"AI-Assisted Data Extraction for Systematic Reviews in Education","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX","json":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX.json","graph_json":"https://pith.science/api/pith-number/6TLMA5MESRVURSOB3CDBJJSCIX/graph.json","events_json":"https://pith.science/api/pith-number/6TLMA5MESRVURSOB3CDBJJSCIX/events.json","paper":"https://pith.science/paper/6TLMA5ME"},"agent_actions":{"view_html":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX","download_json":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX.json","view_paper":"https://pith.science/paper/6TLMA5ME","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14445&json=true","fetch_graph":"https://pith.science/api/pith-number/6TLMA5MESRVURSOB3CDBJJSCIX/graph.json","fetch_events":"https://pith.science/api/pith-number/6TLMA5MESRVURSOB3CDBJJSCIX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX/action/storage_attestation","attest_author":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX/action/author_attestation","sign_citation":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX/action/citation_signature","submit_replication":"https://pith.science/pith/6TLMA5MESRVURSOB3CDBJJSCIX/action/replication_record"}},"created_at":"2026-07-05T10:13:40.181611+00:00","updated_at":"2026-07-05T10:13:40.181611+00:00"}