{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FRFVPXYKUOZ2NITPXI3V5EHKTV","short_pith_number":"pith:FRFVPXYK","schema_version":"1.0","canonical_sha256":"2c4b57df0aa3b3a6a26fba375e90ea9d7f6e724aa6f20c6b46f496861c4e0e58","source":{"kind":"arxiv","id":"2306.04675","version":2},"attestation_state":"computed","paper":{"title":"Exposing flaws of generative model evaluation metrics and their unfair treatment of diffusion models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anthony L. Caterini, Brendan Leigh Ross, Gabriel Loaiza-Ganem, George Stein, J. Eric T. Taylor, Jesse C. Cresswell, Rasa Hosseinzadeh, Valentin Villecroze, Yi Sui, Zhaoyan Liu","submitted_at":"2023-06-07T18:00:00Z","abstract_excerpt":"We systematically study a wide variety of generative models spanning semantically-diverse image datasets to understand and improve the feature extractors and metrics used to evaluate them. Using best practices in psychophysics, we measure human perception of image realism for generated samples by conducting the largest experiment evaluating generative models to date, and find that no existing metric strongly correlates with human evaluations. Comparing to 17 modern metrics for evaluating the overall performance, fidelity, diversity, rarity, and memorization of generative models, we find that t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.04675","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-07T18:00:00Z","cross_cats_sorted":["cs.CV","stat.ML"],"title_canon_sha256":"7b6019babb8f5b2516c340995af82493bf1ffddf18534b35afa98ddbeec82484","abstract_canon_sha256":"0a6a222536a9e204c90adf901bed3611fad8ee10702d528081070301dba5916d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:20:39.648901Z","signature_b64":"PQi2XAIc8Zg5ViPTTaKCLYcnZZ6lqP4bYEnJJGZ62YAumMJ+criiT4vXA2eq6D0W4ku156huxM6AxEtIMgPDAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c4b57df0aa3b3a6a26fba375e90ea9d7f6e724aa6f20c6b46f496861c4e0e58","last_reissued_at":"2026-07-05T07:20:39.648354Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:20:39.648354Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exposing flaws of generative model evaluation metrics and their unfair treatment of diffusion models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anthony L. Caterini, Brendan Leigh Ross, Gabriel Loaiza-Ganem, George Stein, J. Eric T. Taylor, Jesse C. Cresswell, Rasa Hosseinzadeh, Valentin Villecroze, Yi Sui, Zhaoyan Liu","submitted_at":"2023-06-07T18:00:00Z","abstract_excerpt":"We systematically study a wide variety of generative models spanning semantically-diverse image datasets to understand and improve the feature extractors and metrics used to evaluate them. Using best practices in psychophysics, we measure human perception of image realism for generated samples by conducting the largest experiment evaluating generative models to date, and find that no existing metric strongly correlates with human evaluations. Comparing to 17 modern metrics for evaluating the overall performance, fidelity, diversity, rarity, and memorization of generative models, we find that t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.04675","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.04675/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.04675","created_at":"2026-07-05T07:20:39.648418+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.04675v2","created_at":"2026-07-05T07:20:39.648418+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.04675","created_at":"2026-07-05T07:20:39.648418+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRFVPXYKUOZ2","created_at":"2026-07-05T07:20:39.648418+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRFVPXYKUOZ2NITP","created_at":"2026-07-05T07:20:39.648418+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRFVPXYK","created_at":"2026-07-05T07:20:39.648418+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.13419","citing_title":"Diffusion Models Memorize in Training -- and Generalize in Inference","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08976","citing_title":"Score-Based Generative Modeling through Anisotropic Stochastic Partial Differential Equations","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV","json":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV.json","graph_json":"https://pith.science/api/pith-number/FRFVPXYKUOZ2NITPXI3V5EHKTV/graph.json","events_json":"https://pith.science/api/pith-number/FRFVPXYKUOZ2NITPXI3V5EHKTV/events.json","paper":"https://pith.science/paper/FRFVPXYK"},"agent_actions":{"view_html":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV","download_json":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV.json","view_paper":"https://pith.science/paper/FRFVPXYK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.04675&json=true","fetch_graph":"https://pith.science/api/pith-number/FRFVPXYKUOZ2NITPXI3V5EHKTV/graph.json","fetch_events":"https://pith.science/api/pith-number/FRFVPXYKUOZ2NITPXI3V5EHKTV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV/action/storage_attestation","attest_author":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV/action/author_attestation","sign_citation":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV/action/citation_signature","submit_replication":"https://pith.science/pith/FRFVPXYKUOZ2NITPXI3V5EHKTV/action/replication_record"}},"created_at":"2026-07-05T07:20:39.648418+00:00","updated_at":"2026-07-05T07:20:39.648418+00:00"}