{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DDZ5IX2HGE7FG3LTW2BRGOVHSN","short_pith_number":"pith:DDZ5IX2H","schema_version":"1.0","canonical_sha256":"18f3d45f47313e536d73b683133aa79370a87a8c3bc25cab9a18d1f8fa2bc62a","source":{"kind":"arxiv","id":"2309.02726","version":3},"attestation_state":"computed","paper":{"title":"Large Language Models for Automated Open-domain Scientific Hypotheses Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Erik Cambria, Jie Zheng, Junxian Li, Soujanya Poria, Xinya Du, Zonglin Yang","submitted_at":"2023-09-06T05:19:41Z","abstract_excerpt":"Hypothetical induction is recognized as the main reasoning type when scientists make observations about the world and try to propose hypotheses to explain those observations. Past research on hypothetical induction is under a constrained setting: (1) the observation annotations in the dataset are carefully manually handpicked sentences (resulting in a close-domain setting); and (2) the ground truth hypotheses are mostly commonsense knowledge, making the task less challenging. In this work, we tackle these problems by proposing the first dataset for social science academic hypotheses discovery,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.02726","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-06T05:19:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8cddac4a1bda6b4b14bb1b25707213cca1ab456af16332ac4be76c21f6cf88bb","abstract_canon_sha256":"7ee7083e32aff2d904266fe941438d54394cb282c06404d6b640e34584b6aee3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:29.638565Z","signature_b64":"wo5YDD4VTaKAEdNQrtg483gm8sv7pzX/5LGLisW+yzB9YCWVQTtYbCcPay22E2I5L/wksY7NpNyGVmOOy/mhBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18f3d45f47313e536d73b683133aa79370a87a8c3bc25cab9a18d1f8fa2bc62a","last_reissued_at":"2026-07-05T08:30:29.638043Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:29.638043Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models for Automated Open-domain Scientific Hypotheses Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Erik Cambria, Jie Zheng, Junxian Li, Soujanya Poria, Xinya Du, Zonglin Yang","submitted_at":"2023-09-06T05:19:41Z","abstract_excerpt":"Hypothetical induction is recognized as the main reasoning type when scientists make observations about the world and try to propose hypotheses to explain those observations. Past research on hypothetical induction is under a constrained setting: (1) the observation annotations in the dataset are carefully manually handpicked sentences (resulting in a close-domain setting); and (2) the ground truth hypotheses are mostly commonsense knowledge, making the task less challenging. In this work, we tackle these problems by proposing the first dataset for social science academic hypotheses discovery,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.02726","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.02726/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.02726","created_at":"2026-07-05T08:30:29.638109+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.02726v3","created_at":"2026-07-05T08:30:29.638109+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.02726","created_at":"2026-07-05T08:30:29.638109+00:00"},{"alias_kind":"pith_short_12","alias_value":"DDZ5IX2HGE7F","created_at":"2026-07-05T08:30:29.638109+00:00"},{"alias_kind":"pith_short_16","alias_value":"DDZ5IX2HGE7FG3LT","created_at":"2026-07-05T08:30:29.638109+00:00"},{"alias_kind":"pith_short_8","alias_value":"DDZ5IX2H","created_at":"2026-07-05T08:30:29.638109+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22610","citing_title":"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21228","citing_title":"Sakana Fugu Technical Report","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19349","citing_title":"ShinkaEvolve: Towards Open-Ended And Sample-Efficient Program Evolution","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11258","citing_title":"Unlocking LLM Creativity in Science through Analogical Reasoning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2408.06292","citing_title":"The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery","ref_index":109,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN","json":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN.json","graph_json":"https://pith.science/api/pith-number/DDZ5IX2HGE7FG3LTW2BRGOVHSN/graph.json","events_json":"https://pith.science/api/pith-number/DDZ5IX2HGE7FG3LTW2BRGOVHSN/events.json","paper":"https://pith.science/paper/DDZ5IX2H"},"agent_actions":{"view_html":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN","download_json":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN.json","view_paper":"https://pith.science/paper/DDZ5IX2H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.02726&json=true","fetch_graph":"https://pith.science/api/pith-number/DDZ5IX2HGE7FG3LTW2BRGOVHSN/graph.json","fetch_events":"https://pith.science/api/pith-number/DDZ5IX2HGE7FG3LTW2BRGOVHSN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN/action/storage_attestation","attest_author":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN/action/author_attestation","sign_citation":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN/action/citation_signature","submit_replication":"https://pith.science/pith/DDZ5IX2HGE7FG3LTW2BRGOVHSN/action/replication_record"}},"created_at":"2026-07-05T08:30:29.638109+00:00","updated_at":"2026-07-05T08:30:29.638109+00:00"}