{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:7I3RDU4NOX32G2RFFLJIPE7FAS","short_pith_number":"pith:7I3RDU4N","schema_version":"1.0","canonical_sha256":"fa3711d38d75f7a36a252ad28793e50487882761d3db50c3d20f2173e8272a76","source":{"kind":"arxiv","id":"2002.04513","version":2},"attestation_state":"computed","paper":{"title":"An experiment exploring the theoretical and methodological challenges in developing a semi-automated approach to analysis of small-N qualitative data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Sandro Tsang","submitted_at":"2020-02-03T17:55:19Z","abstract_excerpt":"This paper experiments with designing a semi-automated qualitative data analysis (QDA) algorithm to analyse 20 transcripts by using freeware. Text-mining (TM) and QDA were guided by frequency and association measures, because these statistics remain robust when the sample size is small. The refined TM algorithm split the text into various sizes based on a manually revised dictionary. This lemmatisation approach may reflect the context of the text better than uniformly tokenising the text into one single size. TM results were used for initial coding. Code repacking was guided by association mea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.04513","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2020-02-03T17:55:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"b18dc7d7bda793387f1aa1b80d1e414b28ebc0a092debc49dfedd984c6bfc2a3","abstract_canon_sha256":"3ec987c31f418a48d0bbd98594cd6ff05ac099975a354b6df4c14de738fc0792"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:41:03.717237Z","signature_b64":"Tt0TKJvJgPS7He+AFwxPOZFP+jbJ6C5vH+Yu+MMXF2i8Qn8Dx70ze7Bat9R+OeWr0rKPCeA9T4Xj6HYsq7YBCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa3711d38d75f7a36a252ad28793e50487882761d3db50c3d20f2173e8272a76","last_reissued_at":"2026-07-05T00:41:03.716848Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:41:03.716848Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An experiment exploring the theoretical and methodological challenges in developing a semi-automated approach to analysis of small-N qualitative data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Sandro Tsang","submitted_at":"2020-02-03T17:55:19Z","abstract_excerpt":"This paper experiments with designing a semi-automated qualitative data analysis (QDA) algorithm to analyse 20 transcripts by using freeware. Text-mining (TM) and QDA were guided by frequency and association measures, because these statistics remain robust when the sample size is small. The refined TM algorithm split the text into various sizes based on a manually revised dictionary. This lemmatisation approach may reflect the context of the text better than uniformly tokenising the text into one single size. TM results were used for initial coding. Code repacking was guided by association mea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.04513","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.04513/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.04513","created_at":"2026-07-05T00:41:03.716905+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.04513v2","created_at":"2026-07-05T00:41:03.716905+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.04513","created_at":"2026-07-05T00:41:03.716905+00:00"},{"alias_kind":"pith_short_12","alias_value":"7I3RDU4NOX32","created_at":"2026-07-05T00:41:03.716905+00:00"},{"alias_kind":"pith_short_16","alias_value":"7I3RDU4NOX32G2RF","created_at":"2026-07-05T00:41:03.716905+00:00"},{"alias_kind":"pith_short_8","alias_value":"7I3RDU4N","created_at":"2026-07-05T00:41:03.716905+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.19384","citing_title":"From Inductive to Deductive: LLMs-Based Qualitative Data Analysis in Requirements Engineering","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS","json":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS.json","graph_json":"https://pith.science/api/pith-number/7I3RDU4NOX32G2RFFLJIPE7FAS/graph.json","events_json":"https://pith.science/api/pith-number/7I3RDU4NOX32G2RFFLJIPE7FAS/events.json","paper":"https://pith.science/paper/7I3RDU4N"},"agent_actions":{"view_html":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS","download_json":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS.json","view_paper":"https://pith.science/paper/7I3RDU4N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.04513&json=true","fetch_graph":"https://pith.science/api/pith-number/7I3RDU4NOX32G2RFFLJIPE7FAS/graph.json","fetch_events":"https://pith.science/api/pith-number/7I3RDU4NOX32G2RFFLJIPE7FAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS/action/storage_attestation","attest_author":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS/action/author_attestation","sign_citation":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS/action/citation_signature","submit_replication":"https://pith.science/pith/7I3RDU4NOX32G2RFFLJIPE7FAS/action/replication_record"}},"created_at":"2026-07-05T00:41:03.716905+00:00","updated_at":"2026-07-05T00:41:03.716905+00:00"}