{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BLJU46PPK6XJCRL22UNVZRWF2Z","short_pith_number":"pith:BLJU46PP","schema_version":"1.0","canonical_sha256":"0ad34e79ef57ae91457ad51b5cc6c5d66828e32fa8ac4ea39fde3c6d8280a8ca","source":{"kind":"arxiv","id":"2506.11117","version":1},"attestation_state":"computed","paper":{"title":"ScIRGen: Synthesize Realistic and Large-Scale RAG Dataset for Scientific Research","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Hao Liu, Hui Xiong, Junyong Lin, Lu Dai, Min Feng, Qinglin Wu, Ruilin Wang, Ruiqian Han, Xingliang Sun, Yijie Sui","submitted_at":"2025-06-09T11:47:13Z","abstract_excerpt":"Scientific researchers need intensive information about datasets to effectively evaluate and develop theories and methodologies. The information needs regarding datasets are implicitly embedded in particular research tasks, rather than explicitly expressed in search queries. However, existing scientific retrieval and question-answering (QA) datasets typically address straightforward questions, which do not align with the distribution of real-world research inquiries. To bridge this gap, we developed ScIRGen, a dataset generation framework for scientific QA \\& retrieval that more accurately ref"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.11117","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-09T11:47:13Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"fdee450a79dbba0ad0a5d5a29374e7a2e2393e95c545ce20c0a29af9b724aeef","abstract_canon_sha256":"19591a80ef1bc01d946154d6829494cec6d0d07ae68175b93890184a3b6c9509"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:59.776101Z","signature_b64":"66tHdHFmhzheo4zM2RpNHGs7oCquEWZUqQgaVKgFl4Vvu92g+EIeEX0KToFpQxnaRf9yG+gJVTKh7DTssQweBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ad34e79ef57ae91457ad51b5cc6c5d66828e32fa8ac4ea39fde3c6d8280a8ca","last_reissued_at":"2026-07-05T11:20:59.775587Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:59.775587Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ScIRGen: Synthesize Realistic and Large-Scale RAG Dataset for Scientific Research","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Hao Liu, Hui Xiong, Junyong Lin, Lu Dai, Min Feng, Qinglin Wu, Ruilin Wang, Ruiqian Han, Xingliang Sun, Yijie Sui","submitted_at":"2025-06-09T11:47:13Z","abstract_excerpt":"Scientific researchers need intensive information about datasets to effectively evaluate and develop theories and methodologies. The information needs regarding datasets are implicitly embedded in particular research tasks, rather than explicitly expressed in search queries. However, existing scientific retrieval and question-answering (QA) datasets typically address straightforward questions, which do not align with the distribution of real-world research inquiries. To bridge this gap, we developed ScIRGen, a dataset generation framework for scientific QA \\& retrieval that more accurately ref"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11117","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.11117/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.11117","created_at":"2026-07-05T11:20:59.775685+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.11117v1","created_at":"2026-07-05T11:20:59.775685+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11117","created_at":"2026-07-05T11:20:59.775685+00:00"},{"alias_kind":"pith_short_12","alias_value":"BLJU46PPK6XJ","created_at":"2026-07-05T11:20:59.775685+00:00"},{"alias_kind":"pith_short_16","alias_value":"BLJU46PPK6XJCRL2","created_at":"2026-07-05T11:20:59.775685+00:00"},{"alias_kind":"pith_short_8","alias_value":"BLJU46PP","created_at":"2026-07-05T11:20:59.775685+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z","json":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z.json","graph_json":"https://pith.science/api/pith-number/BLJU46PPK6XJCRL22UNVZRWF2Z/graph.json","events_json":"https://pith.science/api/pith-number/BLJU46PPK6XJCRL22UNVZRWF2Z/events.json","paper":"https://pith.science/paper/BLJU46PP"},"agent_actions":{"view_html":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z","download_json":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z.json","view_paper":"https://pith.science/paper/BLJU46PP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.11117&json=true","fetch_graph":"https://pith.science/api/pith-number/BLJU46PPK6XJCRL22UNVZRWF2Z/graph.json","fetch_events":"https://pith.science/api/pith-number/BLJU46PPK6XJCRL22UNVZRWF2Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z/action/storage_attestation","attest_author":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z/action/author_attestation","sign_citation":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z/action/citation_signature","submit_replication":"https://pith.science/pith/BLJU46PPK6XJCRL22UNVZRWF2Z/action/replication_record"}},"created_at":"2026-07-05T11:20:59.775685+00:00","updated_at":"2026-07-05T11:20:59.775685+00:00"}