{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TENPPA3VQVYU5ZFHLBIMDW7WRW","short_pith_number":"pith:TENPPA3V","schema_version":"1.0","canonical_sha256":"991af7837585714ee4a75850c1dbf68dbb2470e8132eeb61f14304bb3a4455eb","source":{"kind":"arxiv","id":"2505.24615","version":1},"attestation_state":"computed","paper":{"title":"Harnessing Large Language Models for Scientific Novelty Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Erik Cambria, Soujanya Poria, Thanh-Son Nguyen, Yan Liu, Zonglin Yang","submitted_at":"2025-05-30T14:08:13Z","abstract_excerpt":"In an era of exponential scientific growth, identifying novel research ideas is crucial and challenging in academia. Despite potential, the lack of an appropriate benchmark dataset hinders the research of novelty detection. More importantly, simply adopting existing NLP technologies, e.g., retrieving and then cross-checking, is not a one-size-fits-all solution due to the gap between textual similarity and idea conception. In this paper, we propose to harness large language models (LLMs) for scientific novelty detection (ND), associated with two new datasets in marketing and NLP domains. To con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24615","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-30T14:08:13Z","cross_cats_sorted":[],"title_canon_sha256":"387fe4c7d92bab54cf5db221576a4170e71b2c66bde7108b24fd1c5a1b5b3a2a","abstract_canon_sha256":"52605d87aba3443d28786dbe3ca8e01436205c9e19cf5f35579e87f30cd9497d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:53.383611Z","signature_b64":"j0vdvNY7ZrhVt7VeUalPUP/STGhFymkUNSc3r4Y167m4esUgNSJhG+gFAQIpWeYZBFmDLngimuTfPy0rgUvSDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"991af7837585714ee4a75850c1dbf68dbb2470e8132eeb61f14304bb3a4455eb","last_reissued_at":"2026-07-05T11:12:53.383100Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:53.383100Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Harnessing Large Language Models for Scientific Novelty Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Erik Cambria, Soujanya Poria, Thanh-Son Nguyen, Yan Liu, Zonglin Yang","submitted_at":"2025-05-30T14:08:13Z","abstract_excerpt":"In an era of exponential scientific growth, identifying novel research ideas is crucial and challenging in academia. Despite potential, the lack of an appropriate benchmark dataset hinders the research of novelty detection. More importantly, simply adopting existing NLP technologies, e.g., retrieving and then cross-checking, is not a one-size-fits-all solution due to the gap between textual similarity and idea conception. In this paper, we propose to harness large language models (LLMs) for scientific novelty detection (ND), associated with two new datasets in marketing and NLP domains. To con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24615","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24615/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24615","created_at":"2026-07-05T11:12:53.383163+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24615v1","created_at":"2026-07-05T11:12:53.383163+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24615","created_at":"2026-07-05T11:12:53.383163+00:00"},{"alias_kind":"pith_short_12","alias_value":"TENPPA3VQVYU","created_at":"2026-07-05T11:12:53.383163+00:00"},{"alias_kind":"pith_short_16","alias_value":"TENPPA3VQVYU5ZFH","created_at":"2026-07-05T11:12:53.383163+00:00"},{"alias_kind":"pith_short_8","alias_value":"TENPPA3V","created_at":"2026-07-05T11:12:53.383163+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.11810","citing_title":"Evolving Roles of LLMs in Scientific Innovation: Assistant, Collaborator, Scientist, and Evaluator","ref_index":106,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW","json":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW.json","graph_json":"https://pith.science/api/pith-number/TENPPA3VQVYU5ZFHLBIMDW7WRW/graph.json","events_json":"https://pith.science/api/pith-number/TENPPA3VQVYU5ZFHLBIMDW7WRW/events.json","paper":"https://pith.science/paper/TENPPA3V"},"agent_actions":{"view_html":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW","download_json":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW.json","view_paper":"https://pith.science/paper/TENPPA3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24615&json=true","fetch_graph":"https://pith.science/api/pith-number/TENPPA3VQVYU5ZFHLBIMDW7WRW/graph.json","fetch_events":"https://pith.science/api/pith-number/TENPPA3VQVYU5ZFHLBIMDW7WRW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW/action/storage_attestation","attest_author":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW/action/author_attestation","sign_citation":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW/action/citation_signature","submit_replication":"https://pith.science/pith/TENPPA3VQVYU5ZFHLBIMDW7WRW/action/replication_record"}},"created_at":"2026-07-05T11:12:53.383163+00:00","updated_at":"2026-07-05T11:12:53.383163+00:00"}