{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UVGLM6SNWMJH7MRLYN7E6ONCLV","short_pith_number":"pith:UVGLM6SN","schema_version":"1.0","canonical_sha256":"a54cb67a4db3127fb22bc37e4f39a25d6907b3403fe0890fff3f5437024186d3","source":{"kind":"arxiv","id":"2504.09620","version":1},"attestation_state":"computed","paper":{"title":"Metropolis-Hastings Captioning Game: Knowledge Fusion of Vision Language Models via Decentralized Bayesian Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.MA"],"primary_cat":"cs.CL","authors_text":"Ryosuke Yamaki, Ryo Ueda, Seitaro Shinagawa, Tadahiro Taniguchi, Yuta Matsui","submitted_at":"2025-04-13T15:28:09Z","abstract_excerpt":"We propose the Metropolis-Hastings Captioning Game (MHCG), a method to fuse knowledge of multiple vision-language models (VLMs) by learning from each other. Although existing methods that combine multiple models suffer from inference costs and architectural constraints, MHCG avoids these problems by performing decentralized Bayesian inference through a process resembling a language game. The knowledge fusion process establishes communication between two VLM agents alternately captioning images and learning from each other. We conduct two image-captioning experiments with two VLMs, each pre-tra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.09620","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-13T15:28:09Z","cross_cats_sorted":["cs.AI","cs.CV","cs.MA"],"title_canon_sha256":"5ddf4a90e6b96b3f3236fc55860c5dd2c5d1885359617ff0b9d9dfacec2b4a1e","abstract_canon_sha256":"e248370e422ea7a3138cf8c09436bb76f549865875d0fca1796af8b6a65cbb1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:40.140146Z","signature_b64":"SdcCR4cvK6GwYBYCvr5H+pICYyyoZBgtXAcJOhb+zPL4KoTMh27F2DDlF84rIk/nXDR+n1kr9WQZJYiX1OqBCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a54cb67a4db3127fb22bc37e4f39a25d6907b3403fe0890fff3f5437024186d3","last_reissued_at":"2026-07-05T10:48:40.139613Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:40.139613Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Metropolis-Hastings Captioning Game: Knowledge Fusion of Vision Language Models via Decentralized Bayesian Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.MA"],"primary_cat":"cs.CL","authors_text":"Ryosuke Yamaki, Ryo Ueda, Seitaro Shinagawa, Tadahiro Taniguchi, Yuta Matsui","submitted_at":"2025-04-13T15:28:09Z","abstract_excerpt":"We propose the Metropolis-Hastings Captioning Game (MHCG), a method to fuse knowledge of multiple vision-language models (VLMs) by learning from each other. Although existing methods that combine multiple models suffer from inference costs and architectural constraints, MHCG avoids these problems by performing decentralized Bayesian inference through a process resembling a language game. The knowledge fusion process establishes communication between two VLM agents alternately captioning images and learning from each other. We conduct two image-captioning experiments with two VLMs, each pre-tra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.09620","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.09620/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.09620","created_at":"2026-07-05T10:48:40.139668+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.09620v1","created_at":"2026-07-05T10:48:40.139668+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.09620","created_at":"2026-07-05T10:48:40.139668+00:00"},{"alias_kind":"pith_short_12","alias_value":"UVGLM6SNWMJH","created_at":"2026-07-05T10:48:40.139668+00:00"},{"alias_kind":"pith_short_16","alias_value":"UVGLM6SNWMJH7MRL","created_at":"2026-07-05T10:48:40.139668+00:00"},{"alias_kind":"pith_short_8","alias_value":"UVGLM6SN","created_at":"2026-07-05T10:48:40.139668+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11695","citing_title":"Emergent Communication between Heterogeneous Visual Agents through Decentralized Learning","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV","json":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV.json","graph_json":"https://pith.science/api/pith-number/UVGLM6SNWMJH7MRLYN7E6ONCLV/graph.json","events_json":"https://pith.science/api/pith-number/UVGLM6SNWMJH7MRLYN7E6ONCLV/events.json","paper":"https://pith.science/paper/UVGLM6SN"},"agent_actions":{"view_html":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV","download_json":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV.json","view_paper":"https://pith.science/paper/UVGLM6SN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.09620&json=true","fetch_graph":"https://pith.science/api/pith-number/UVGLM6SNWMJH7MRLYN7E6ONCLV/graph.json","fetch_events":"https://pith.science/api/pith-number/UVGLM6SNWMJH7MRLYN7E6ONCLV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV/action/storage_attestation","attest_author":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV/action/author_attestation","sign_citation":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV/action/citation_signature","submit_replication":"https://pith.science/pith/UVGLM6SNWMJH7MRLYN7E6ONCLV/action/replication_record"}},"created_at":"2026-07-05T10:48:40.139668+00:00","updated_at":"2026-07-05T10:48:40.139668+00:00"}