{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K6QUU6MLAL6KJCPSDAA6EY5LYJ","short_pith_number":"pith:K6QUU6ML","schema_version":"1.0","canonical_sha256":"57a14a798b02fca489f21801e263abc27fe3f61213e5db8925aa62472c9f728d","source":{"kind":"arxiv","id":"2406.11280","version":2},"attestation_state":"computed","paper":{"title":"ISR-DPO: Aligning Large Multimodal Models for Videos by Iterative Self-Retrospective DPO","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daechul Ahn, Dongyeop Kang, Jonghyun Choi, San Kim, Youngjae Yu, Yura Choi","submitted_at":"2024-06-17T07:33:30Z","abstract_excerpt":"Iterative self-improvement, a concept extending beyond personal growth, has found powerful applications in machine learning, particularly in transforming weak models into strong ones. While recent advances in natural language processing have shown its efficacy through iterative preference optimization, applying this approach to Video Large Multi-modal Models (VLMMs) remains challenging due to modality misalignment. VLMMs struggle with this misalignment during iterative preference modeling, as the self-judge model often prioritizes linguistic knowledge over visual information. Additionally, ite"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11280","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-17T07:33:30Z","cross_cats_sorted":[],"title_canon_sha256":"557753bf978e2cf045c3b9be708c72e4ddbff31ca3d490760e9091ed16253af8","abstract_canon_sha256":"f86a4351b6083e4fde28004fcbc3f79e7e39eecc83d2ce68bd24625fe1b2b382"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:21.570241Z","signature_b64":"/85IKRz9CyleufM8DdfHbFI8poc6mSvnFapG/Lue9y4pIy1oFUHYT4GAfOtyAjNLaOy3SxhFHj9BawnDp8thDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57a14a798b02fca489f21801e263abc27fe3f61213e5db8925aa62472c9f728d","last_reissued_at":"2026-07-05T09:58:21.569695Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:21.569695Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ISR-DPO: Aligning Large Multimodal Models for Videos by Iterative Self-Retrospective DPO","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Daechul Ahn, Dongyeop Kang, Jonghyun Choi, San Kim, Youngjae Yu, Yura Choi","submitted_at":"2024-06-17T07:33:30Z","abstract_excerpt":"Iterative self-improvement, a concept extending beyond personal growth, has found powerful applications in machine learning, particularly in transforming weak models into strong ones. While recent advances in natural language processing have shown its efficacy through iterative preference optimization, applying this approach to Video Large Multi-modal Models (VLMMs) remains challenging due to modality misalignment. VLMMs struggle with this misalignment during iterative preference modeling, as the self-judge model often prioritizes linguistic knowledge over visual information. Additionally, ite"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11280","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11280","created_at":"2026-07-05T09:58:21.569756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11280v2","created_at":"2026-07-05T09:58:21.569756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11280","created_at":"2026-07-05T09:58:21.569756+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6QUU6MLAL6K","created_at":"2026-07-05T09:58:21.569756+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6QUU6MLAL6KJCPS","created_at":"2026-07-05T09:58:21.569756+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6QUU6ML","created_at":"2026-07-05T09:58:21.569756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.02984","citing_title":"From Answers to Rationales: Self-Aligning Multimodal Reasoning with Answer-Oriented Chain-of-Thought","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ","json":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ.json","graph_json":"https://pith.science/api/pith-number/K6QUU6MLAL6KJCPSDAA6EY5LYJ/graph.json","events_json":"https://pith.science/api/pith-number/K6QUU6MLAL6KJCPSDAA6EY5LYJ/events.json","paper":"https://pith.science/paper/K6QUU6ML"},"agent_actions":{"view_html":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ","download_json":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ.json","view_paper":"https://pith.science/paper/K6QUU6ML","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11280&json=true","fetch_graph":"https://pith.science/api/pith-number/K6QUU6MLAL6KJCPSDAA6EY5LYJ/graph.json","fetch_events":"https://pith.science/api/pith-number/K6QUU6MLAL6KJCPSDAA6EY5LYJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ/action/storage_attestation","attest_author":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ/action/author_attestation","sign_citation":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ/action/citation_signature","submit_replication":"https://pith.science/pith/K6QUU6MLAL6KJCPSDAA6EY5LYJ/action/replication_record"}},"created_at":"2026-07-05T09:58:21.569756+00:00","updated_at":"2026-07-05T09:58:21.569756+00:00"}