{"schema_version":2,"identifier":"https://tovaki.com/research/ai-video-model-benchmark-methodology#methodology","protocol_id":"tovaki-ai-video-blind-benchmark-2026-r1","canonical_url":"https://tovaki.com/research/ai-video-model-benchmark-methodology","machine_readable_url":"https://tovaki.com/data/ai-video-model-benchmark-methodology.json","agent_readable_url":"https://tovaki.com/data/ai-video-model-benchmark-methodology.md","publisher":{"name":"Tovaki","url":"https://tovaki.com"},"status":"prepared_not_executed","published_at":"2026-08-25","reviewed_at":"2026-08-25","evidence_boundary":"This record describes a planned Tovaki integration benchmark. No provider generation, human review, quality result or model ranking is represented as complete.","protocol_commitment":{"algorithm":"SHA-256","digest":"3efa03ed5fb305054069571bee501e0a06cb247e3334530c9dd7820cff8840d7","covers":"The complete private protocol file, including its exact prompts, model roster, scoring weights, retry policy and publication gate.","interpretation":"A matching digest can prove that a later disclosed protocol is byte-for-byte identical to the version committed here; the digest does not prove that the benchmark was executed."},"fixed_scope":{"model_integrations":4,"task_lanes":6,"first_take_cells":24,"reviewers":3,"duration_seconds":6,"aspect_ratio":"9:16","tovaki_resolution_tier":"720p","audio_condition":"Native synchronized audio expected for every accepted sample; optional native audio is enabled and always-on audio remains enabled.","review_proxy":"720x1280 H.264 at 24 fps with AAC 48 kHz audio, normalized with one locked transcode recipe while originals are preserved."},"task_lanes":["relationship_handoff","dialogue_reaction","moving_action_geography","material_contact_and_recovery","continuous_camera_and_identity","bright_surreal_causality"],"scoring":{"scale":"0-5 in 0.5 increments","weighted_total":100,"dimensions":[{"key":"prompt_adherence","weight":15},{"key":"temporal_stability","weight":15},{"key":"subject_identity","weight":15},{"key":"motion_contact","weight":15},{"key":"camera_geography","weight":10},{"key":"performance_reaction","weight":10},{"key":"sound_completion","weight":10},{"key":"usable_edit_rate","weight":10}],"tie_threshold_points_out_of_100":3,"disagreement_flag_range_on_0_to_5_scale":3},"controls":{"same_exact_prompt_and_settings_per_task":true,"task_local_anonymous_labels":"A-D, independently randomized inside each task","reviewer_specific_order":true,"model_and_vendor_identity_hidden_until_score_lock":true,"exact_media_hash_required":"SHA-256 for every accepted review proxy","direct_audio_monitoring_required":true,"scorecards_hash_locked_before_reveal":3,"explicit_unlock_required_to_reveal_identity":true},"retry_policy":{"maximum_retries_per_cell":1,"creative_retry_forbidden":true,"allowed_reasons":["provider_error","timeout","corrupt_or_unplayable_file","wrong_duration_or_resolution_from_provider","missing_required_audio_track"],"selection_note":"A visually weak, surprising or unattractive result is evidence, not a reason to reroll."},"budget_gate":{"currency":"Tovaki credits","first_pass_planned_credits":1620,"hard_maximum_with_one_technical_retry_per_cell":3240,"generation_authorized":false,"approved_limit_credits":null,"named_owner_approval_required_before_generation":true},"disclosure":{"public_now":["scope and fixed settings","task-lane coverage without exact prompts","score dimensions and weights","randomization, retry, locking and publication rules","protocol SHA-256 commitment"],"withheld_until_scores_are_locked":["exact prompts","model roster for this round","task-local anonymous mapping"],"publish_with_any_result":["exact prompts and settings","accepted sample hashes","technical failures and retries with sensitive request identifiers redacted","anonymous raw reviewer scores","actual Tovaki credits and limitations"],"remain_private":["reviewer personal identities unless each reviewer consents","provider credentials and request tokens","private provider metadata not needed to reproduce the result"]},"publication_gate":{"ranking_available":false,"results_available":false,"universal_best_claim_allowed":false,"required_scope_language":"in this six-task Tovaki integration sample","close_result_rule":"If the top two aggregate scores differ by less than 3 points out of 100, report that they are not distinguishable in this sample."},"methodological_references":[{"title":"VBench: Comprehensive Benchmark Suite for Video Generative Models","url":"https://openaccess.thecvf.com/content/CVPR2024/papers/Huang_VBench_Comprehensive_Benchmark_Suite_for_Video_Generative_Models_CVPR_2024_paper.pdf","used_for":"Separating temporal quality, frame-wise quality and condition consistency rather than treating a polished frame as complete video quality."},{"title":"TC-Bench: Benchmarking Temporal Compositionality in Text-to-Video and Image-to-Video Generation","url":"https://arxiv.org/abs/2406.08656","used_for":"Testing whether object attributes and relations change in the requested order while identities remain traceable through time."},{"title":"Video-Bench: Human-Aligned Video Generation Benchmark","url":"https://openaccess.thecvf.com/content/CVPR2025/papers/Han_Video-Bench_Human-Aligned_Video_Generation_Benchmark_CVPR_2025_paper.pdf","used_for":"Keeping human judgment attached to explicit dimensions and publishing the limits of an aggregate score."},{"title":"Artificial Analysis video generation benchmarking methodology","url":"https://artificialanalysis.ai/video/methodology","used_for":"Treating audio-enabled generation as its own comparison condition and declaring fixed settings and endpoint scope."}]}