{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:XKMQ5MDUMZRKU2QXH2PLUFIBLS","short_pith_number":"pith:XKMQ5MDU","schema_version":"1.0","canonical_sha256":"ba990eb0746662aa6a173e9eba15015cad13112d061269f1347203cab05f2469","source":{"kind":"arxiv","id":"2605.22819","version":1},"attestation_state":"computed","paper":{"title":"Cambrian-P: Pose-Grounded Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Hu Xu, Jihan Yang, Junyi Zhang, Saining Xie, Shusheng Yang, Xichen Pan, Zifan Zhao","submitted_at":"2026-05-21T17:59:45Z","abstract_excerpt":"Camera pose matters. The position and orientation of each viewpoint define a shared spatial coordinate frame that relates observations across video frames. Yet this signal is largely absent from multimodal LLMs (MLLMs) for video understanding, which process frames as isolated 2D snapshots, instead of the persistent scene humans perceive. We revisit pose as a lightweight supervisory signal and introduce Cambrian-P, a video MLLM augmented with per-frame learnable camera tokens and a pose regression head. With a carefully designed sampling scheme, the model achieves substantial gains of 4.5-6.5% "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2605.22819","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2026-05-21T17:59:45Z","cross_cats_sorted":[],"title_canon_sha256":"e314e3801a61b5dd35cc39a29d41c01559783605b56bdd7a0d5b80fe61d77cd1","abstract_canon_sha256":"4c1a461a828066f0d73669f1ee076d327896d925bc6d2f9157bcc986ddc25c14"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-22T02:04:56.293439Z","signature_b64":"5BwPlRBNSILdyXWDbkRYG5xTVEYMZ3gKyfWWg2VHJClhRJbZVmzxNQdVQyLBOOFfXAGpvvfDsiRjz5x1YmneAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba990eb0746662aa6a173e9eba15015cad13112d061269f1347203cab05f2469","last_reissued_at":"2026-05-22T02:04:56.292964Z","signature_status":"signed_v1","first_computed_at":"2026-05-22T02:04:56.292964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cambrian-P: Pose-Grounded Video Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Hu Xu, Jihan Yang, Junyi Zhang, Saining Xie, Shusheng Yang, Xichen Pan, Zifan Zhao","submitted_at":"2026-05-21T17:59:45Z","abstract_excerpt":"Camera pose matters. The position and orientation of each viewpoint define a shared spatial coordinate frame that relates observations across video frames. Yet this signal is largely absent from multimodal LLMs (MLLMs) for video understanding, which process frames as isolated 2D snapshots, instead of the persistent scene humans perceive. We revisit pose as a lightweight supervisory signal and introduce Cambrian-P, a video MLLM augmented with per-frame learnable camera tokens and a pose regression head. With a carefully designed sampling scheme, the model achieves substantial gains of 4.5-6.5% "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.22819","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.22819/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2605.22819","created_at":"2026-05-22T02:04:56.293032+00:00"},{"alias_kind":"arxiv_version","alias_value":"2605.22819v1","created_at":"2026-05-22T02:04:56.293032+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.22819","created_at":"2026-05-22T02:04:56.293032+00:00"},{"alias_kind":"pith_short_12","alias_value":"XKMQ5MDUMZRK","created_at":"2026-05-22T02:04:56.293032+00:00"},{"alias_kind":"pith_short_16","alias_value":"XKMQ5MDUMZRKU2QX","created_at":"2026-05-22T02:04:56.293032+00:00"},{"alias_kind":"pith_short_8","alias_value":"XKMQ5MDU","created_at":"2026-05-22T02:04:56.293032+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS","json":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS.json","graph_json":"https://pith.science/api/pith-number/XKMQ5MDUMZRKU2QXH2PLUFIBLS/graph.json","events_json":"https://pith.science/api/pith-number/XKMQ5MDUMZRKU2QXH2PLUFIBLS/events.json","paper":"https://pith.science/paper/XKMQ5MDU"},"agent_actions":{"view_html":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS","download_json":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS.json","view_paper":"https://pith.science/paper/XKMQ5MDU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2605.22819&json=true","fetch_graph":"https://pith.science/api/pith-number/XKMQ5MDUMZRKU2QXH2PLUFIBLS/graph.json","fetch_events":"https://pith.science/api/pith-number/XKMQ5MDUMZRKU2QXH2PLUFIBLS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS/action/storage_attestation","attest_author":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS/action/author_attestation","sign_citation":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS/action/citation_signature","submit_replication":"https://pith.science/pith/XKMQ5MDUMZRKU2QXH2PLUFIBLS/action/replication_record"}},"created_at":"2026-05-22T02:04:56.293032+00:00","updated_at":"2026-05-22T02:04:56.293032+00:00"}