{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:5EPRZFV2XU65SQGYOCRQ77LWZH","short_pith_number":"pith:5EPRZFV2","schema_version":"1.0","canonical_sha256":"e91f1c96babd3dd940d870a30ffd76c9e6c4c618a154972adfca83423b797a9f","source":{"kind":"arxiv","id":"2607.28516","version":1},"attestation_state":"computed","paper":{"title":"Beyond Frame Selection: Generative Latent Evidence Aggregation for Long-Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bodong Du, Bowen Liu, Shuning Wang, Xiaomeng Li, Xinpeng Ding, Zhiheng Wu","submitted_at":"2026-07-30T16:55:00Z","abstract_excerpt":"Long-video understanding commonly compresses videos into a small set of frames or visual tokens for answer generation. Existing compact pipelines focus on retaining relevant visual content as explicit evidence. Yet making evidence available does not ensure that complementary cues across moments are integrated for answering. Our key idea is to organize selected frames into query-relevant cross-frame evidence before generation. We formulate this post-selection stage as a latent evidence interface and instantiate it with GenEvA ($\\textbf{Gen}erative$ $Latent$ $\\textbf{Ev}idence$ $\\textbf{A}ggrega"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.28516","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-07-30T16:55:00Z","cross_cats_sorted":[],"title_canon_sha256":"cd600d7fe594ffaa17fda1905822e67246e13ba7dfbab77700f9c54c0fde547d","abstract_canon_sha256":"74f16ce63fb770a95eb7ec079b0e3becb173415a00daa457ba520428b899fff7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e91f1c96babd3dd940d870a30ffd76c9e6c4c618a154972adfca83423b797a9f","last_reissued_at":"2026-07-31T01:38:22.718048Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-31T01:38:22.718048Z"},"graph_snapshot":{"paper":{"title":"Beyond Frame Selection: Generative Latent Evidence Aggregation for Long-Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bodong Du, Bowen Liu, Shuning Wang, Xiaomeng Li, Xinpeng Ding, Zhiheng Wu","submitted_at":"2026-07-30T16:55:00Z","abstract_excerpt":"Long-video understanding commonly compresses videos into a small set of frames or visual tokens for answer generation. Existing compact pipelines focus on retaining relevant visual content as explicit evidence. Yet making evidence available does not ensure that complementary cues across moments are integrated for answering. Our key idea is to organize selected frames into query-relevant cross-frame evidence before generation. We formulate this post-selection stage as a latent evidence interface and instantiate it with GenEvA ($\\textbf{Gen}erative$ $Latent$ $\\textbf{Ev}idence$ $\\textbf{A}ggrega"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.28516","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.28516/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.28516","created_at":"2026-07-31T01:38:22.721198+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.28516v1","created_at":"2026-07-31T01:38:22.721198+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.28516","created_at":"2026-07-31T01:38:22.721198+00:00"},{"alias_kind":"pith_short_12","alias_value":"5EPRZFV2XU65","created_at":"2026-07-31T01:38:22.721198+00:00"},{"alias_kind":"pith_short_16","alias_value":"5EPRZFV2XU65SQGY","created_at":"2026-07-31T01:38:22.721198+00:00"},{"alias_kind":"pith_short_8","alias_value":"5EPRZFV2","created_at":"2026-07-31T01:38:22.721198+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH","json":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH.json","graph_json":"https://pith.science/api/pith-number/5EPRZFV2XU65SQGYOCRQ77LWZH/graph.json","events_json":"https://pith.science/api/pith-number/5EPRZFV2XU65SQGYOCRQ77LWZH/events.json","paper":"https://pith.science/paper/5EPRZFV2"},"agent_actions":{"view_html":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH","download_json":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH.json","view_paper":"https://pith.science/paper/5EPRZFV2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.28516&json=true","fetch_graph":"https://pith.science/api/pith-number/5EPRZFV2XU65SQGYOCRQ77LWZH/graph.json","fetch_events":"https://pith.science/api/pith-number/5EPRZFV2XU65SQGYOCRQ77LWZH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH/action/storage_attestation","attest_author":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH/action/author_attestation","sign_citation":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH/action/citation_signature","submit_replication":"https://pith.science/pith/5EPRZFV2XU65SQGYOCRQ77LWZH/action/replication_record"}},"created_at":"2026-07-31T01:38:22.721198+00:00","updated_at":"2026-07-31T01:38:22.721198+00:00"}