{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DHAV5EQGGVKWUITTSSASWX4CMP","short_pith_number":"pith:DHAV5EQG","schema_version":"1.0","canonical_sha256":"19c15e920635556a227394812b5f8263ecc3bcd1ffad1af2180cda558a4c446c","source":{"kind":"arxiv","id":"2502.11181","version":1},"attestation_state":"computed","paper":{"title":"Improving Scientific Document Retrieval with Concept Coverage-based Query Set Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Bowen Jin, Dongha Lee, HwanJo Yu, Jiawei Han, SeongKu Kang, Wonbin Kweon, Yu Zhang","submitted_at":"2025-02-16T15:59:50Z","abstract_excerpt":"In specialized fields like the scientific domain, constructing large-scale human-annotated datasets poses a significant challenge due to the need for domain expertise. Recent methods have employed large language models to generate synthetic queries, which serve as proxies for actual user queries. However, they lack control over the content generated, often resulting in incomplete coverage of academic concepts in documents. We introduce Concept Coverage-based Query set Generation (CCQGen) framework, designed to generate a set of queries with comprehensive coverage of the document's concepts. A "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11181","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.IR","submitted_at":"2025-02-16T15:59:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"dabc6911bf2abf0ac7641bc7692154791e0b800354b91779aecdaaee751a8662","abstract_canon_sha256":"f822ae07808a7bc27cd70fe958c87e6b9cc7a5d95e63f59819bf8789269c1074"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:04.749381Z","signature_b64":"4AeqyXfn/dOp13VxzOuHvrGJFyLE7ZDLQhFmwOtnDxrO0OfAPp6DpYkGDKz0uZnB56QQ8ehmQFluH7Xz9mg9DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19c15e920635556a227394812b5f8263ecc3bcd1ffad1af2180cda558a4c446c","last_reissued_at":"2026-07-05T10:15:04.748896Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:04.748896Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Scientific Document Retrieval with Concept Coverage-based Query Set Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Bowen Jin, Dongha Lee, HwanJo Yu, Jiawei Han, SeongKu Kang, Wonbin Kweon, Yu Zhang","submitted_at":"2025-02-16T15:59:50Z","abstract_excerpt":"In specialized fields like the scientific domain, constructing large-scale human-annotated datasets poses a significant challenge due to the need for domain expertise. Recent methods have employed large language models to generate synthetic queries, which serve as proxies for actual user queries. However, they lack control over the content generated, often resulting in incomplete coverage of academic concepts in documents. We introduce Concept Coverage-based Query set Generation (CCQGen) framework, designed to generate a set of queries with comprehensive coverage of the document's concepts. A "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11181","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11181/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11181","created_at":"2026-07-05T10:15:04.748954+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11181v1","created_at":"2026-07-05T10:15:04.748954+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11181","created_at":"2026-07-05T10:15:04.748954+00:00"},{"alias_kind":"pith_short_12","alias_value":"DHAV5EQGGVKW","created_at":"2026-07-05T10:15:04.748954+00:00"},{"alias_kind":"pith_short_16","alias_value":"DHAV5EQGGVKWUITT","created_at":"2026-07-05T10:15:04.748954+00:00"},{"alias_kind":"pith_short_8","alias_value":"DHAV5EQG","created_at":"2026-07-05T10:15:04.748954+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.13757","citing_title":"CoRank: LLM-Based Compact Reranking with Document Features for Scientific Retrieval","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP","json":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP.json","graph_json":"https://pith.science/api/pith-number/DHAV5EQGGVKWUITTSSASWX4CMP/graph.json","events_json":"https://pith.science/api/pith-number/DHAV5EQGGVKWUITTSSASWX4CMP/events.json","paper":"https://pith.science/paper/DHAV5EQG"},"agent_actions":{"view_html":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP","download_json":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP.json","view_paper":"https://pith.science/paper/DHAV5EQG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11181&json=true","fetch_graph":"https://pith.science/api/pith-number/DHAV5EQGGVKWUITTSSASWX4CMP/graph.json","fetch_events":"https://pith.science/api/pith-number/DHAV5EQGGVKWUITTSSASWX4CMP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP/action/storage_attestation","attest_author":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP/action/author_attestation","sign_citation":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP/action/citation_signature","submit_replication":"https://pith.science/pith/DHAV5EQGGVKWUITTSSASWX4CMP/action/replication_record"}},"created_at":"2026-07-05T10:15:04.748954+00:00","updated_at":"2026-07-05T10:15:04.748954+00:00"}