{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BM2IPIOGVIO6PFNOSYUUXFOHFN","short_pith_number":"pith:BM2IPIOG","schema_version":"1.0","canonical_sha256":"0b3487a1c6aa1de795ae96294b95c72b79957d5d2c40aadbd399abc6ae0d2b29","source":{"kind":"arxiv","id":"2502.17540","version":1},"attestation_state":"computed","paper":{"title":"PosterSum: A Multimodal Benchmark for Scientific Poster Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Frank Keller, Pasquale Minervini, Rohit Saxena","submitted_at":"2025-02-24T18:35:39Z","abstract_excerpt":"Generating accurate and concise textual summaries from multimodal documents is challenging, especially when dealing with visually complex content like scientific posters. We introduce PosterSum, a novel benchmark to advance the development of vision-language models that can understand and summarize scientific posters into research paper abstracts. Our dataset contains 16,305 conference posters paired with their corresponding abstracts as summaries. Each poster is provided in image format and presents diverse visual understanding challenges, such as complex layouts, dense text regions, tables, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17540","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-24T18:35:39Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"3e95d0fc49f63befb1c79a05670d3cc78b3423740d56d5f0995ef60c5b02ea8a","abstract_canon_sha256":"c3c19933d777b055616dcda79639cc324672bb0c151daf82fb4a8d70adcf43c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:12.344009Z","signature_b64":"x7UID6AudOdWMd+duVKsbKy19WRe43E44NaFSAdywJ3oH8NQyxgGU/Q+YFN15NZsP2N4KclMpGJ/dO1Xd1OFCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b3487a1c6aa1de795ae96294b95c72b79957d5d2c40aadbd399abc6ae0d2b29","last_reissued_at":"2026-07-05T10:19:12.343518Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:12.343518Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PosterSum: A Multimodal Benchmark for Scientific Poster Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Frank Keller, Pasquale Minervini, Rohit Saxena","submitted_at":"2025-02-24T18:35:39Z","abstract_excerpt":"Generating accurate and concise textual summaries from multimodal documents is challenging, especially when dealing with visually complex content like scientific posters. We introduce PosterSum, a novel benchmark to advance the development of vision-language models that can understand and summarize scientific posters into research paper abstracts. Our dataset contains 16,305 conference posters paired with their corresponding abstracts as summaries. Each poster is provided in image format and presents diverse visual understanding challenges, such as complex layouts, dense text regions, tables, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17540","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17540/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17540","created_at":"2026-07-05T10:19:12.343578+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17540v1","created_at":"2026-07-05T10:19:12.343578+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17540","created_at":"2026-07-05T10:19:12.343578+00:00"},{"alias_kind":"pith_short_12","alias_value":"BM2IPIOGVIO6","created_at":"2026-07-05T10:19:12.343578+00:00"},{"alias_kind":"pith_short_16","alias_value":"BM2IPIOGVIO6PFNO","created_at":"2026-07-05T10:19:12.343578+00:00"},{"alias_kind":"pith_short_8","alias_value":"BM2IPIOG","created_at":"2026-07-05T10:19:12.343578+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02915","citing_title":"Any2Poster: Any-Source Poster Generation Across Modalities and Domains","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21720","citing_title":"PosterForest: Hierarchical Multi-Agent Collaboration for Scientific Poster Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11253","citing_title":"VideoAgent: Personalized Synthesis of Scientific Videos","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22490","citing_title":"SciPostGen: Bridging the Gap between Scientific Papers and Poster Layouts","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08980","citing_title":"Training Multi-Image Vision Agents via End2End Reinforcement Learning","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN","json":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN.json","graph_json":"https://pith.science/api/pith-number/BM2IPIOGVIO6PFNOSYUUXFOHFN/graph.json","events_json":"https://pith.science/api/pith-number/BM2IPIOGVIO6PFNOSYUUXFOHFN/events.json","paper":"https://pith.science/paper/BM2IPIOG"},"agent_actions":{"view_html":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN","download_json":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN.json","view_paper":"https://pith.science/paper/BM2IPIOG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17540&json=true","fetch_graph":"https://pith.science/api/pith-number/BM2IPIOGVIO6PFNOSYUUXFOHFN/graph.json","fetch_events":"https://pith.science/api/pith-number/BM2IPIOGVIO6PFNOSYUUXFOHFN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN/action/storage_attestation","attest_author":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN/action/author_attestation","sign_citation":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN/action/citation_signature","submit_replication":"https://pith.science/pith/BM2IPIOGVIO6PFNOSYUUXFOHFN/action/replication_record"}},"created_at":"2026-07-05T10:19:12.343578+00:00","updated_at":"2026-07-05T10:19:12.343578+00:00"}