{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:DPQ22QU65CCC35XGTUC3WCVQJP","short_pith_number":"pith:DPQ22QU6","schema_version":"1.0","canonical_sha256":"1be1ad429ee8842df6e69d05bb0ab04bf8a1b317054782e7c1918eaaed45c728","source":{"kind":"arxiv","id":"2604.15145","version":2},"attestation_state":"computed","paper":{"title":"An Axiomatic Benchmark for Evaluation of Scientific Novelty Metrics","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"No single metric satisfies all axioms for scientific novelty, but combining complementary architectures reaches 90.1 percent compliance.","cross_cats":["cs.DL"],"primary_cat":"cs.AI","authors_text":"ChengXiang Zhai, Miri Liu","submitted_at":"2026-04-16T15:19:58Z","abstract_excerpt":"The rigorous evaluation of the novelty of a scientific paper is, even for human scientists, a challenging task. With the increasing interest in AI scientists, it is becoming more and more important that this task be automatable and reliable, lest attention and compute be wasted on ideas that have already been explored. Due to the challenge of quantifying ground-truth novelty, however, existing novelty metrics generally validate against noisy, confounded signals such as citation counts or peer review scores. We introduce a benchmark that compares novelty metrics without requiring explicit novel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2604.15145","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2026-04-16T15:19:58Z","cross_cats_sorted":["cs.DL"],"title_canon_sha256":"216e4c00bf4aa30bae8db2a97ac360162f870f288d5668cc136ced87ce5f7431","abstract_canon_sha256":"c6a6a475293d0e53d672b0cb62644f844b6d05901d4f06aef1d4d2aaa8155c6c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-07T00:45:50.715232Z","signature_b64":"3RYw6rhMgDajKZFS5ClU1weKQvMoajD8TpSuOrOpez5SKcUsn631HZoicnzW0SysTi6+Sz8C9dkTavLR9YsvBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1be1ad429ee8842df6e69d05bb0ab04bf8a1b317054782e7c1918eaaed45c728","last_reissued_at":"2026-08-07T00:45:50.713682Z","signature_status":"signed_v1","first_computed_at":"2026-08-07T00:45:50.713682Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Axiomatic Benchmark for Evaluation of Scientific Novelty Metrics","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"No single metric satisfies all axioms for scientific novelty, but combining complementary architectures reaches 90.1 percent compliance.","cross_cats":["cs.DL"],"primary_cat":"cs.AI","authors_text":"ChengXiang Zhai, Miri Liu","submitted_at":"2026-04-16T15:19:58Z","abstract_excerpt":"The rigorous evaluation of the novelty of a scientific paper is, even for human scientists, a challenging task. With the increasing interest in AI scientists, it is becoming more and more important that this task be automatable and reliable, lest attention and compute be wasted on ideas that have already been explored. Due to the challenge of quantifying ground-truth novelty, however, existing novelty metrics generally validate against noisy, confounded signals such as citation counts or peer review scores. We introduce a benchmark that compares novelty metrics without requiring explicit novel"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Combining metrics of complementary architectures leads to consistent improvements on the benchmark, with per-axiom weighting achieving 90.1% versus 71.5% for the best individual metric.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The axioms defined capture the essential aspects of human scientific norms for novelty, and the ten tasks across three AI domains sufficiently represent the general problem of evaluating scientific novelty.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"An axiomatic benchmark shows no single novelty metric satisfies all desired properties consistently, but combining complementary metrics reaches 90.1% performance versus 71.5% for the best individual metric.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"No single metric satisfies all axioms for scientific novelty, but combining complementary architectures reaches 90.1 percent compliance.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"0bd77ca5a717a571e0dbc5ed10056c40c33d29a6869f64c4ed61b2fa1394a84c"},"source":{"id":"2604.15145","kind":"arxiv","version":2},"verdict":{"id":"7c0dd367-2a57-4add-8b43-173d079d7342","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T11:18:46.444284Z","strongest_claim":"Combining metrics of complementary architectures leads to consistent improvements on the benchmark, with per-axiom weighting achieving 90.1% versus 71.5% for the best individual metric.","one_line_summary":"An axiomatic benchmark shows no single novelty metric satisfies all desired properties consistently, but combining complementary metrics reaches 90.1% performance versus 71.5% for the best individual metric.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The axioms defined capture the essential aspects of human scientific norms for novelty, and the ten tasks across three AI domains sufficiently represent the general problem of evaluating scientific novelty.","pith_extraction_headline":"No single metric satisfies all axioms for scientific novelty, but combining complementary architectures reaches 90.1 percent compliance."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.15145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2604.15145","created_at":"2026-08-07T00:45:50.714993+00:00"},{"alias_kind":"arxiv_version","alias_value":"2604.15145v2","created_at":"2026-08-07T00:45:50.714993+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.15145","created_at":"2026-08-07T00:45:50.714993+00:00"},{"alias_kind":"pith_short_12","alias_value":"DPQ22QU65CCC","created_at":"2026-08-07T00:45:50.714993+00:00"},{"alias_kind":"pith_short_16","alias_value":"DPQ22QU65CCC35XG","created_at":"2026-08-07T00:45:50.714993+00:00"},{"alias_kind":"pith_short_8","alias_value":"DPQ22QU6","created_at":"2026-08-07T00:45:50.714993+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP","json":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP.json","graph_json":"https://pith.science/api/pith-number/DPQ22QU65CCC35XGTUC3WCVQJP/graph.json","events_json":"https://pith.science/api/pith-number/DPQ22QU65CCC35XGTUC3WCVQJP/events.json","paper":"https://pith.science/paper/DPQ22QU6"},"agent_actions":{"view_html":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP","download_json":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP.json","view_paper":"https://pith.science/paper/DPQ22QU6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2604.15145&json=true","fetch_graph":"https://pith.science/api/pith-number/DPQ22QU65CCC35XGTUC3WCVQJP/graph.json","fetch_events":"https://pith.science/api/pith-number/DPQ22QU65CCC35XGTUC3WCVQJP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP/action/storage_attestation","attest_author":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP/action/author_attestation","sign_citation":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP/action/citation_signature","submit_replication":"https://pith.science/pith/DPQ22QU65CCC35XGTUC3WCVQJP/action/replication_record"}},"created_at":"2026-08-07T00:45:50.714993+00:00","updated_at":"2026-08-07T00:45:50.714993+00:00"}