{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4TMHJBS4U765WMDWLHXIZSHZ7F","short_pith_number":"pith:4TMHJBS4","schema_version":"1.0","canonical_sha256":"e4d874865ca7fddb307659ee8cc8f9f963f20d2bfe34c01e35019557fc940089","source":{"kind":"arxiv","id":"2404.07824","version":1},"attestation_state":"computed","paper":{"title":"Heron-Bench: A Benchmark for Evaluating Vision Language Models in Japanese","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Kazuki Fujii, Kento Sasaki, Kotaro Tanahashi, Yuichi Inoue, Yuma Ochi, Yu Yamaguchi","submitted_at":"2024-04-11T15:09:22Z","abstract_excerpt":"Vision Language Models (VLMs) have undergone a rapid evolution, giving rise to significant advancements in the realm of multimodal understanding tasks. However, the majority of these models are trained and evaluated on English-centric datasets, leaving a gap in the development and evaluation of VLMs for other languages, such as Japanese. This gap can be attributed to the lack of methodologies for constructing VLMs and the absence of benchmarks to accurately measure their performance. To address this issue, we introduce a novel benchmark, Japanese Heron-Bench, for evaluating Japanese capabiliti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.07824","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-11T15:09:22Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"986bf82784edbade07939ab834940aa003baa6c02a4641f1f3b32b01ad5382ed","abstract_canon_sha256":"10dc333088690eb0232dd8d9572ffb2802578ba7b4ee3ed9effd7a9a23747b66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:58.231273Z","signature_b64":"7alq5POfiviz3rh2nimgMH7w99CMRgq6jN62mGouoaPvIkN53gQaoKAaQnKJueahTTqEVs/vF/CSzUHYDHDuCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4d874865ca7fddb307659ee8cc8f9f963f20d2bfe34c01e35019557fc940089","last_reissued_at":"2026-07-05T08:06:58.230813Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:58.230813Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Heron-Bench: A Benchmark for Evaluating Vision Language Models in Japanese","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Kazuki Fujii, Kento Sasaki, Kotaro Tanahashi, Yuichi Inoue, Yuma Ochi, Yu Yamaguchi","submitted_at":"2024-04-11T15:09:22Z","abstract_excerpt":"Vision Language Models (VLMs) have undergone a rapid evolution, giving rise to significant advancements in the realm of multimodal understanding tasks. However, the majority of these models are trained and evaluated on English-centric datasets, leaving a gap in the development and evaluation of VLMs for other languages, such as Japanese. This gap can be attributed to the lack of methodologies for constructing VLMs and the absence of benchmarks to accurately measure their performance. To address this issue, we introduce a novel benchmark, Japanese Heron-Bench, for evaluating Japanese capabiliti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.07824","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.07824/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.07824","created_at":"2026-07-05T08:06:58.230872+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.07824v1","created_at":"2026-07-05T08:06:58.230872+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.07824","created_at":"2026-07-05T08:06:58.230872+00:00"},{"alias_kind":"pith_short_12","alias_value":"4TMHJBS4U765","created_at":"2026-07-05T08:06:58.230872+00:00"},{"alias_kind":"pith_short_16","alias_value":"4TMHJBS4U765WMDW","created_at":"2026-07-05T08:06:58.230872+00:00"},{"alias_kind":"pith_short_8","alias_value":"4TMHJBS4","created_at":"2026-07-05T08:06:58.230872+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.00700","citing_title":"Contrasting Cognitive Styles in Vision-Language Models: Holistic Attention in Japanese Versus Analytical Focus in English","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F","json":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F.json","graph_json":"https://pith.science/api/pith-number/4TMHJBS4U765WMDWLHXIZSHZ7F/graph.json","events_json":"https://pith.science/api/pith-number/4TMHJBS4U765WMDWLHXIZSHZ7F/events.json","paper":"https://pith.science/paper/4TMHJBS4"},"agent_actions":{"view_html":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F","download_json":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F.json","view_paper":"https://pith.science/paper/4TMHJBS4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.07824&json=true","fetch_graph":"https://pith.science/api/pith-number/4TMHJBS4U765WMDWLHXIZSHZ7F/graph.json","fetch_events":"https://pith.science/api/pith-number/4TMHJBS4U765WMDWLHXIZSHZ7F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F/action/storage_attestation","attest_author":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F/action/author_attestation","sign_citation":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F/action/citation_signature","submit_replication":"https://pith.science/pith/4TMHJBS4U765WMDWLHXIZSHZ7F/action/replication_record"}},"created_at":"2026-07-05T08:06:58.230872+00:00","updated_at":"2026-07-05T08:06:58.230872+00:00"}