{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CJOVXLLMNHYMYW4NOPTMXJGRXY","short_pith_number":"pith:CJOVXLLM","schema_version":"1.0","canonical_sha256":"125d5bad6c69f0cc5b8d73e6cba4d1be09b7c21db455d612548eedf1353fba7a","source":{"kind":"arxiv","id":"2504.12976","version":1},"attestation_state":"computed","paper":{"title":"Sparks of Science: Hypothesis Generation Using Structured Paper Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Charles O'Neill, Ioana Ciuc\\u{a}, Kevin Schawinski, Mike Walmsley, Roberta R\\u{a}ileanu, Thang Bui, Tirthankar Ghosal","submitted_at":"2025-04-17T14:29:18Z","abstract_excerpt":"Generating novel and creative scientific hypotheses is a cornerstone in achieving Artificial General Intelligence. Large language and reasoning models have the potential to aid in the systematic creation, selection, and validation of scientifically informed hypotheses. However, current foundation models often struggle to produce scientific ideas that are both novel and feasible. One reason is the lack of a dedicated dataset that frames Scientific Hypothesis Generation (SHG) as a Natural Language Generation (NLG) task. In this paper, we introduce HypoGen, the first dataset of approximately 5500"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.12976","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-17T14:29:18Z","cross_cats_sorted":[],"title_canon_sha256":"17682eb807e609c9dc1fb0aeafd2fc0607ea20c685a017d9e3edbd350bf9e6da","abstract_canon_sha256":"849bd26b375c86ab0b4ac658589e900b7e5220e3a860415e7d008978675aa0be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:35.684287Z","signature_b64":"EYgytxpaaj48cYv83ip8MSFbQICJ6d1qRkFQJxHtyUby8/rDonI9P7HXqFw/fPQkHXeBnF6YOXZftdCIEW6HAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"125d5bad6c69f0cc5b8d73e6cba4d1be09b7c21db455d612548eedf1353fba7a","last_reissued_at":"2026-07-05T10:50:35.683803Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:35.683803Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparks of Science: Hypothesis Generation Using Structured Paper Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Charles O'Neill, Ioana Ciuc\\u{a}, Kevin Schawinski, Mike Walmsley, Roberta R\\u{a}ileanu, Thang Bui, Tirthankar Ghosal","submitted_at":"2025-04-17T14:29:18Z","abstract_excerpt":"Generating novel and creative scientific hypotheses is a cornerstone in achieving Artificial General Intelligence. Large language and reasoning models have the potential to aid in the systematic creation, selection, and validation of scientifically informed hypotheses. However, current foundation models often struggle to produce scientific ideas that are both novel and feasible. One reason is the lack of a dedicated dataset that frames Scientific Hypothesis Generation (SHG) as a Natural Language Generation (NLG) task. In this paper, we introduce HypoGen, the first dataset of approximately 5500"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.12976","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.12976/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.12976","created_at":"2026-07-05T10:50:35.683863+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.12976v1","created_at":"2026-07-05T10:50:35.683863+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.12976","created_at":"2026-07-05T10:50:35.683863+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJOVXLLMNHYM","created_at":"2026-07-05T10:50:35.683863+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJOVXLLMNHYMYW4N","created_at":"2026-07-05T10:50:35.683863+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJOVXLLM","created_at":"2026-07-05T10:50:35.683863+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08532","citing_title":"DN-Hypo-Pipeline: An AI-Driven Workflow for Generating Hypotheses using Large Language Models and Scientific Explanations","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY","json":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY.json","graph_json":"https://pith.science/api/pith-number/CJOVXLLMNHYMYW4NOPTMXJGRXY/graph.json","events_json":"https://pith.science/api/pith-number/CJOVXLLMNHYMYW4NOPTMXJGRXY/events.json","paper":"https://pith.science/paper/CJOVXLLM"},"agent_actions":{"view_html":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY","download_json":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY.json","view_paper":"https://pith.science/paper/CJOVXLLM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.12976&json=true","fetch_graph":"https://pith.science/api/pith-number/CJOVXLLMNHYMYW4NOPTMXJGRXY/graph.json","fetch_events":"https://pith.science/api/pith-number/CJOVXLLMNHYMYW4NOPTMXJGRXY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY/action/storage_attestation","attest_author":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY/action/author_attestation","sign_citation":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY/action/citation_signature","submit_replication":"https://pith.science/pith/CJOVXLLMNHYMYW4NOPTMXJGRXY/action/replication_record"}},"created_at":"2026-07-05T10:50:35.683863+00:00","updated_at":"2026-07-05T10:50:35.683863+00:00"}