{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FZOGAFWPKD7PVQNKXZS7H6SUUL","short_pith_number":"pith:FZOGAFWP","schema_version":"1.0","canonical_sha256":"2e5c6016cf50fefac1aabe65f3fa54a2e1d33a3690708c8f6cdbab2efdf2b3d7","source":{"kind":"arxiv","id":"2310.09668","version":1},"attestation_state":"computed","paper":{"title":"Beyond Testers' Biases: Guiding Model Testing with Knowledge Bases using LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Chenyang Yang, Christian K\\\"astner, Grace A. Lewis, Rachel Brower-Sinning, Rishabh Rustogi, Tongshuang Wu","submitted_at":"2023-10-14T21:24:03Z","abstract_excerpt":"Current model testing work has mostly focused on creating test cases. Identifying what to test is a step that is largely ignored and poorly supported. We propose Weaver, an interactive tool that supports requirements elicitation for guiding model testing. Weaver uses large language models to generate knowledge bases and recommends concepts from them interactively, allowing testers to elicit requirements for further testing. Weaver provides rich external knowledge to testers and encourages testers to systematically explore diverse concepts beyond their own biases. In a user study, we show that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.09668","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-14T21:24:03Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"369b8c6dc16df40c83d8e2785d8943036cb5f8462dd840773dafb0bdd5c681ec","abstract_canon_sha256":"1982c6857479d5f7f4b547fa629c3ac58dd1cc36c5d6348712cef1a20b0dc9b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:09.447107Z","signature_b64":"pxASRRV5SPIvzYv35pxw/BqwVP/yKe9IYlilRgEBTVx291rwyLGNhMyfQDE3tjXN3Ys0wYqqWoqwwvUq55lNBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e5c6016cf50fefac1aabe65f3fa54a2e1d33a3690708c8f6cdbab2efdf2b3d7","last_reissued_at":"2026-07-05T07:01:09.446636Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:09.446636Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Testers' Biases: Guiding Model Testing with Knowledge Bases using LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Chenyang Yang, Christian K\\\"astner, Grace A. Lewis, Rachel Brower-Sinning, Rishabh Rustogi, Tongshuang Wu","submitted_at":"2023-10-14T21:24:03Z","abstract_excerpt":"Current model testing work has mostly focused on creating test cases. Identifying what to test is a step that is largely ignored and poorly supported. We propose Weaver, an interactive tool that supports requirements elicitation for guiding model testing. Weaver uses large language models to generate knowledge bases and recommends concepts from them interactively, allowing testers to elicit requirements for further testing. Weaver provides rich external knowledge to testers and encourages testers to systematically explore diverse concepts beyond their own biases. In a user study, we show that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.09668","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.09668/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.09668","created_at":"2026-07-05T07:01:09.446696+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.09668v1","created_at":"2026-07-05T07:01:09.446696+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.09668","created_at":"2026-07-05T07:01:09.446696+00:00"},{"alias_kind":"pith_short_12","alias_value":"FZOGAFWPKD7P","created_at":"2026-07-05T07:01:09.446696+00:00"},{"alias_kind":"pith_short_16","alias_value":"FZOGAFWPKD7PVQNK","created_at":"2026-07-05T07:01:09.446696+00:00"},{"alias_kind":"pith_short_8","alias_value":"FZOGAFWP","created_at":"2026-07-05T07:01:09.446696+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04185","citing_title":"From Legal Text to Tech Specs: Generative AI's Interpretation of Consent in Privacy Law","ref_index":33,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL","json":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL.json","graph_json":"https://pith.science/api/pith-number/FZOGAFWPKD7PVQNKXZS7H6SUUL/graph.json","events_json":"https://pith.science/api/pith-number/FZOGAFWPKD7PVQNKXZS7H6SUUL/events.json","paper":"https://pith.science/paper/FZOGAFWP"},"agent_actions":{"view_html":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL","download_json":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL.json","view_paper":"https://pith.science/paper/FZOGAFWP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.09668&json=true","fetch_graph":"https://pith.science/api/pith-number/FZOGAFWPKD7PVQNKXZS7H6SUUL/graph.json","fetch_events":"https://pith.science/api/pith-number/FZOGAFWPKD7PVQNKXZS7H6SUUL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL/action/storage_attestation","attest_author":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL/action/author_attestation","sign_citation":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL/action/citation_signature","submit_replication":"https://pith.science/pith/FZOGAFWPKD7PVQNKXZS7H6SUUL/action/replication_record"}},"created_at":"2026-07-05T07:01:09.446696+00:00","updated_at":"2026-07-05T07:01:09.446696+00:00"}