{
  "schema_version": 1,
  "audit_date": "2026-09-14",
  "source_leaderboard_csv_sha256": "e51013fa766cb0c1cc5602f4ea464522f883c384cf1cd0d0d9cdbe2605e1dfcc",
  "source_extended_csv_sha256": "ecebfc02e68b2e2b4229608f987df896ed399d2e0f164075fc3623f3ad300d62",
  "source_manifest_sha256": "b18ceec2867e030d187530200dfac56e82c03fbb6f29b715beb71c043dba0c3b",
  "parquet_sha256": "39f9da7aca9020d79f383953646a5893f09c6f8e5f60433560011280ee987b2d",
  "ordered_annotation_content_sha256": "90ee915016105f6a709f391e8a03a6d0e99bc5c908f945cdf7b80d0cb289e789",
  "preview_ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
  "scorer_sha256": "3d13630e5aae2c8dcd2ae0f60e0c46243b5544762a16fa8ad7aa1f439fa995ff",
  "api_scorer_sha256": "964731a854e16569ec0d464f79b9125721a1c98c727b53e57b967ca73e7cd52b",
  "api_token_f1_per_sample_max_abs_error": {
    "gemini-3.1-pro-preview": 0,
    "gemini-3.6-flash": 0
  },
  "rows": [
    {
      "method": "grt_llava_ov_0_5b",
      "display_name": "GRT (LLaVA-OneVision Qwen2 0.5B)",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.10125,
        "grid_ade": 0.712142440040729,
        "grid_fde": 0.6864884792587348,
        "grid_transition_acc": 0.014857142857142855,
        "token_f1": 0.18128055327025913
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260814_204224_results.json",
      "result_sha256": "1b980db0fc97dd1ebd4791319d52af3e2f89775567db8edaba920b53cd040808",
      "samples_sha256": "fa525bb49ef70e42a8cfdf08a5edf5e8d5418634c81d4532b6a7064a0199f462",
      "model_wrapper": "llava_ov_dense_video",
      "model_args": "pretrained=lmms-lab/llava-onevision-qwen2-0.5b-ov,conv_template=qwen_1_5,model_name=llava_qwen,video_decode_backend=decord,max_frames_num=8,dense_frame_fps=1,clip_duration_sec=10,profiling=True,use_gated_tok=True,use_vision_merge=True,frame_sampling_strategy=uniform,gate_policy=motion,gate_metric=ssim,gate_diff_threshold=0.05,enable_visual_token_pruning=False,prune_mode=off,scene_merge_apply=True,scene_merge_keep_ratio=0.5,scene_merge_require_semantic_match=True,merge_jsd_threshold=0.02,scene_merge_min_tokens_per_frame=16",
      "recorded_git_hash": "62b10af",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 969,
      "exactly_eight_parsed_labels": 584,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.10125,
        "grid_ade": 0.712142440040729,
        "grid_fde": 0.6864884792587348,
        "grid_transition_acc": 0.014857142857142855,
        "token_f1": 0.18128055327025913
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "36572368fba64a992248bb01320e732965ab49e0263e231315a7e7a2b0642cdb",
      "fps_log_rows": 1000,
      "logged_sampled_frame_counts": {
        "2": 254,
        "3": 184,
        "4": 162,
        "5": 93,
        "6": 54,
        "7": 40,
        "8": 213
      },
      "logged_truncated_clips": 148,
      "error_log_markers": 0,
      "protocol_cohort": "grt_legacy_input_mismatch",
      "input_sampling_policy": "1fps_up_to8_frames_first10_seconds",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform_whole_clip",
      "rank_eligible": false,
      "reason": "787/1000 inputs contain fewer than eight frames;148 clips capped at10sec, while references cover whole-clip eight endpoints. No matched GRT ablation.",
      "artifact_integrity_pass": true,
      "model": "GRT (LLaVA-OneVision Qwen2 0.5B)",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "787/1000 inputs contain fewer than eight frames;148 clips capped at10sec, while references cover whole-clip eight endpoints. No matched GRT ablation."
    },
    {
      "method": "qwen3_vl_2b",
      "display_name": "Qwen3-VL 2B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.1035,
        "grid_ade": 0.43897622358804256,
        "grid_fde": 0.4477007280285228,
        "grid_transition_acc": 0.34585714285714286,
        "token_f1": 0.14988653846153846
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260821_215414_results.json",
      "result_sha256": "510a6207964413d9a697ec30c1c22698ba31e734ed2b13e0887fa938de12f5db",
      "samples_sha256": "e12009e1b4f9aa487367c409a0ec9676da793d0e499dfbd9ff357c18d4f0d15d",
      "model_wrapper": "qwen3_vl",
      "model_args": "pretrained=Qwen/Qwen3-VL-2B-Instruct,trust_remote_code=True,device_map=auto,torch_dtype=bfloat16,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 977,
      "exactly_eight_parsed_labels": 974,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.1035,
        "grid_ade": 0.43897622358804256,
        "grid_fde": 0.4477007280285228,
        "grid_transition_acc": 0.34585714285714286,
        "token_f1": 0.14988653846153846
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "9bc80f211ef89124c07f30dc6daed2ab6af15bed4bef7e203968144fed9e084e",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen3-VL 2B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen3_vl_4b",
      "display_name": "Qwen3-VL 4B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.132375,
        "grid_ade": 0.4137622036744973,
        "grid_fde": 0.42381250837750256,
        "grid_transition_acc": 0.824,
        "token_f1": 0.132375
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_020709_results.json",
      "result_sha256": "cc7a281b6da39846360627c23c6c892def0d022340853a0d7053db2d72a03093",
      "samples_sha256": "a3a1cc5f8c5873f875c2ff9db4062325796e2296f6199820572ff9d63cd5a31b",
      "model_wrapper": "qwen3_vl",
      "model_args": "pretrained=Qwen/Qwen3-VL-4B-Instruct,trust_remote_code=True,device_map=auto,torch_dtype=bfloat16,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 1000,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.132375,
        "grid_ade": 0.4137622036744973,
        "grid_fde": 0.42381250837750256,
        "grid_transition_acc": 0.824,
        "token_f1": 0.132375
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "fde916022d497e83c778aaf14c82a8208f314a221dff8c95a171be215f96c50c",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen3-VL 4B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen3_vl_8b",
      "display_name": "Qwen3-VL 8B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.02075,
        "grid_ade": 0.49089100982790895,
        "grid_fde": 0.5064880728375344,
        "grid_transition_acc": 0.823,
        "token_f1": 0.02075
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_060728_results.json",
      "result_sha256": "277d8b578086448ea5c7fca4c9f49461d5395e9c05c3a9c467d4ef220db241d3",
      "samples_sha256": "caa5ee6a8b93d22cdb3051723134a83d957a1a126a6e36ee18962896e29f843d",
      "model_wrapper": "qwen3_vl",
      "model_args": "pretrained=Qwen/Qwen3-VL-8B-Instruct,trust_remote_code=True,device_map=auto,torch_dtype=bfloat16,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 1000,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.02075,
        "grid_ade": 0.49089100982790895,
        "grid_fde": 0.5064880728375344,
        "grid_transition_acc": 0.823,
        "token_f1": 0.02075
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "4a939230ce7b65298418a3c02b2f16354f6235e317c25ac7bd45afa8d30f920e",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen3-VL 8B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen3_vl_32b",
      "display_name": "Qwen3-VL 32B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.1235,
        "grid_ade": 0.3950253602938521,
        "grid_fde": 0.4043133740387499,
        "grid_transition_acc": 0.8101428571428572,
        "token_f1": 0.125625
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_092305_results.json",
      "result_sha256": "19166012c531b48c1699bbd55d9e1dbb11d3e75b04fc73b28bb610f09a04a4d6",
      "samples_sha256": "89e072f275782b9a31da5059990ace1cae410c95de7350ed9f6754ed968eb82c",
      "model_wrapper": "qwen3_vl",
      "model_args": "pretrained=Qwen/Qwen3-VL-32B-Instruct,trust_remote_code=True,device_map=auto,torch_dtype=bfloat16,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 1000,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.1235,
        "grid_ade": 0.3950253602938521,
        "grid_fde": 0.4043133740387499,
        "grid_transition_acc": 0.8101428571428572,
        "token_f1": 0.125625
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "9d8e0eb3561206885f465571e6d653faceda0bd20b2cf12438ac2a23dc6e68dd",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen3-VL 32B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen2_vl_2b",
      "display_name": "Qwen2-VL 2B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.153625,
        "grid_ade": 0.4391509338579707,
        "grid_fde": 0.22697336110361055,
        "grid_transition_acc": 0.24014285714285707,
        "token_f1": 0.23347887488328667
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_125403_results.json",
      "result_sha256": "8810ac74c835ced91875cf441dd15e05cb5a3164bdfd367aa0c7c85c17c55e33",
      "samples_sha256": "c6a3eebe50897163bb76d4de82c2b70a5ae51040b749854676963aff0a6977ff",
      "model_wrapper": "qwen2_vl",
      "model_args": "pretrained=Qwen/Qwen2-VL-2B-Instruct,device_map=auto,max_num_frames=8",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 992,
      "exactly_eight_parsed_labels": 860,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.153625,
        "grid_ade": 0.4391509338579707,
        "grid_fde": 0.22697336110361055,
        "grid_transition_acc": 0.24014285714285707,
        "token_f1": 0.23347887488328667
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "405685af67804da952f9c2047e9acc21df3a54e0472c7f4a4163ea2e61dd3ca2",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen2-VL 2B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen2_5_vl_3b",
      "display_name": "Qwen2.5-VL 3B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.011625,
        "grid_ade": 0.6538012159543,
        "grid_fde": 0.656330290227057,
        "grid_transition_acc": 0.6424285714285713,
        "token_f1": 0.0145
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_132438_results.json",
      "result_sha256": "20c5873fcae149ffed977658d37f6238a306d98802ae56be58453c60d91a8882",
      "samples_sha256": "aee597078b47931736538428b3ab182ed6b15bed7871dfbeb85a779580e60361",
      "model_wrapper": "qwen2_5_vl",
      "model_args": "pretrained=Qwen/Qwen2.5-VL-3B-Instruct,device_map=auto,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 1000,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.011625,
        "grid_ade": 0.6538012159543,
        "grid_fde": 0.656330290227057,
        "grid_transition_acc": 0.6424285714285713,
        "token_f1": 0.0145
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "8d5ed294a5730cb9b8fa3d59fde6691805a69f518118808f9ac428c7bacfbb73",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen2.5-VL 3B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen2_5_vl_7b",
      "display_name": "Qwen2.5-VL 7B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.02925,
        "grid_ade": 0.6271162193588583,
        "grid_fde": 0.632060323494788,
        "grid_transition_acc": 0.7928571428571428,
        "token_f1": 0.029075000000000004
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_145756_results.json",
      "result_sha256": "9f68979163f8af9a3f620cd310a851c6e5996d1ed03aec4b07f7419935d734cb",
      "samples_sha256": "fe61d4afc69210b933312f066e6c58b39b810b8f0c084a797b76bfa9b9269ac2",
      "model_wrapper": "qwen2_5_vl",
      "model_args": "pretrained=Qwen/Qwen2.5-VL-7B-Instruct,device_map=auto,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 767,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.02925,
        "grid_ade": 0.6271162193588583,
        "grid_fde": 0.632060323494788,
        "grid_transition_acc": 0.7928571428571428,
        "token_f1": 0.029075000000000004
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "1b02d1684506c171b8567aaca1ce79b1645348739b7e80cc75aed23d5cf19902",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen2.5-VL 7B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen2_5_vl_32b",
      "display_name": "Qwen2.5-VL 32B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.31625,
        "grid_ade": 0.31187021350306654,
        "grid_fde": 0.3124830699815178,
        "grid_transition_acc": 0.6332857142857143,
        "token_f1": 0.3307326948828312
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_153746_results.json",
      "result_sha256": "21a187651d94a5a15d2ea1a3d01fff0f7775ac114a2eab7f1afe724cc0104843",
      "samples_sha256": "a8cd6ec8f3ac388d7d992696f90385331cf2b719f4b4f22d6c079877f4ed86a5",
      "model_wrapper": "qwen2_5_vl",
      "model_args": "pretrained=Qwen/Qwen2.5-VL-32B-Instruct,device_map=auto,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 964,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.31625,
        "grid_ade": 0.31187021350306654,
        "grid_fde": 0.3124830699815178,
        "grid_transition_acc": 0.6332857142857143,
        "token_f1": 0.3307326948828312
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "0c8289e2f24207c38282574d318c22ef2d89d2cce016a8816dd99047678799bd",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen2.5-VL 32B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen2_5_vl_72b",
      "display_name": "Qwen2.5-VL 72B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.084,
        "grid_ade": 0.4985684131201151,
        "grid_fde": 0.4841290595170691,
        "grid_transition_acc": 0.726,
        "token_f1": 0.096375
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_162820_results.json",
      "result_sha256": "8d64aa89e45dd180b7a10fc0a2cc07f0a7634404e30ecada9632f55c147b756a",
      "samples_sha256": "0c9d61a5dba105fe7e5be6f60cbcbf641272df0ce1ba67db03d05dc9ef0781e7",
      "model_wrapper": "qwen2_5_vl",
      "model_args": "pretrained=Qwen/Qwen2.5-VL-72B-Instruct,device_map=auto,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 1000,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.084,
        "grid_ade": 0.4985684131201151,
        "grid_fde": 0.4841290595170691,
        "grid_transition_acc": 0.726,
        "token_f1": 0.096375
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "7662109b67f1d3de0154a9f6c7041f1aedbaa48a42d297311a873ad1d8662969",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen2.5-VL 72B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "llava_onevision_0_5b",
      "display_name": "LLaVA-OneVision Qwen2 0.5B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.142875,
        "grid_ade": 0.5091069196938063,
        "grid_fde": 0.46643800755695713,
        "grid_transition_acc": 0.01714285714285714,
        "token_f1": 0.21888529905064952
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_172222_results.json",
      "result_sha256": "71419bd0e9d9b2e52a99ebf267bc6fa4fdec82733eb59a120847d176e2c1ed41",
      "samples_sha256": "f9616d1effa3bf94112be04d5ce9eb090b702a346662ffd43aaaa7d68aa08bd6",
      "model_wrapper": "llava_hf",
      "model_args": "pretrained=llava-hf/llava-onevision-qwen2-0.5b-ov-hf,trust_remote_code=True,device_map=auto,dtype=bfloat16,max_frames_num=8,max_image_size=384,attn_implementation=eager,profiling=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 945,
      "exactly_eight_parsed_labels": 23,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.142875,
        "grid_ade": 0.5091069196938063,
        "grid_fde": 0.46643800755695713,
        "grid_transition_acc": 0.01714285714285714,
        "token_f1": 0.21888529905064952
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "f7f6f798fa0db102a3948611efe3db351cb3cebd0f0f3e908510a988dd5cd796",
      "fps_log_rows": 1000,
      "logged_sampled_frame_counts": {
        "8": 1000
      },
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "LLaVA-OneVision Qwen2 0.5B",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "llava_onevision_original",
      "display_name": "LLaVA-OneVision Qwen2 7B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.137375,
        "grid_ade": 0.44693965102762345,
        "grid_fde": 0.44719316784359886,
        "grid_transition_acc": 0.2171428571428571,
        "token_f1": 0.223255264833206
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_181427_results.json",
      "result_sha256": "c50a57f728830baf1804afb1b22545dff2bacf2eb2b6a6e3586acfadc326f9ce",
      "samples_sha256": "ea700324136e6574463ee7cb46d995726dc8b1e37eeca5d752ec2de5a680a5ab",
      "model_wrapper": "llava_hf",
      "model_args": "pretrained=llava-hf/llava-onevision-qwen2-7b-ov-hf,trust_remote_code=True,device_map=auto,dtype=bfloat16,max_frames_num=8,max_image_size=384,attn_implementation=eager",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 748,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.137375,
        "grid_ade": 0.44693965102762345,
        "grid_fde": 0.44719316784359886,
        "grid_transition_acc": 0.2171428571428571,
        "token_f1": 0.223255264833206
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "4887d044f75fbe5d6d373ce61f972ff38b742c8df0515971f52a439014f3d720",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "LLaVA-OneVision Qwen2 7B",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "qwen2_vl_7b",
      "display_name": "Qwen2-VL 7B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.225875,
        "grid_ade": 0.3465179492345876,
        "grid_fde": 0.3160468098071215,
        "grid_transition_acc": 0.5097142857142857,
        "token_f1": 0.26896743697479
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260822_233713_results.json",
      "result_sha256": "bee7dd3ca4c4e8fa7485d5213ff52b057ab5a66448ef9dc2e6e6a118d1596a4e",
      "samples_sha256": "c3e371971e56eb63c09f5fa6ad328ce4c2d8adce30572f7eb8cf492e6f2dc64a",
      "model_wrapper": "qwen2_vl",
      "model_args": "pretrained=Qwen/Qwen2-VL-7B-Instruct,device_map=auto,max_num_frames=8,max_image_size=384,use_custom_video_loader=True",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 988,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.225875,
        "grid_ade": 0.3465179492345876,
        "grid_fde": 0.3160468098071215,
        "grid_transition_acc": 0.5097142857142857,
        "token_f1": 0.26896743697479
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "070b972f786adc754dd3ed55c4c934e66b8dde53ea8dac0c01d7adcee377230b",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Qwen2-VL 7B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "llava_onevision_1_5_8b",
      "display_name": "LLaVA-OneVision 1.5 8B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.02175,
        "grid_ade": 0.5041977500138574,
        "grid_fde": 0.5237004583432506,
        "grid_transition_acc": 0.8235714285714284,
        "token_f1": 0.022033333333333332
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_015806_results.json",
      "result_sha256": "227ff7fc1e66f6fa512328a4b52121589552eaaff568123a8f8d28c711a4701d",
      "samples_sha256": "48f30a494a409cca0b1268c4c4eceba7382b8110e6f6234e7aab08719f8807fb",
      "model_wrapper": "llava_hf",
      "model_args": "pretrained=lmms-lab/LLaVA-OneVision-1.5-8B-Instruct,trust_remote_code=True,device_map=auto,dtype=bfloat16,max_frames_num=8,max_image_size=384,attn_implementation=eager",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 998,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.02175,
        "grid_ade": 0.5041977500138574,
        "grid_fde": 0.5237004583432506,
        "grid_transition_acc": 0.8235714285714284,
        "token_f1": 0.022033333333333332
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "1386d97288bf5801f941be101ddbc2a34cf519aba0c5158b926a041baa190bab",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "LLaVA-OneVision 1.5 8B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "llava_onevision_2_8b",
      "display_name": "LLaVA-OneVision 2 8B Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.02225,
        "grid_ade": 0.44972882469420067,
        "grid_fde": 0.4566500369747288,
        "grid_transition_acc": 0.8174285714285713,
        "token_f1": 0.022354166666666668
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_093108_results.json",
      "result_sha256": "79cafec85162fec41f0ba44e44064528dc8a1d2b506329b0357e474ccc5cbb11",
      "samples_sha256": "af4229d6a9013cb8c4809a217ac6ce964f60ad76a58dcb432221ecd1561c0a31",
      "model_wrapper": "llava_hf",
      "model_args": "pretrained=lmms-lab-encoder/LLaVA-OneVision-2-8B-Instruct,trust_remote_code=True,device_map=auto,dtype=bfloat16,max_frames_num=8,max_image_size=384,attn_implementation=eager",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 998,
      "exactly_eight_parsed_labels": 994,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.02225,
        "grid_ade": 0.44972882469420067,
        "grid_fde": 0.4566500369747288,
        "grid_transition_acc": 0.8174285714285713,
        "token_f1": 0.022354166666666668
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "9ccd840076547e85741b90f06ef077401f3975ebcc9ee4a3921afb91808a8512",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "LLaVA-OneVision 2 8B Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "internvl2_5_1b",
      "display_name": "InternVL2.5 1B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.082625,
        "grid_ade": 0.7740200084651281,
        "grid_fde": 1.2456937564708708,
        "grid_transition_acc": 0.025857142857142853,
        "token_f1": 0.16582469715689713
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_102001_results.json",
      "result_sha256": "67c0f5d901600cdfcad37d63c16ef26c246265a5d9087e42ac6f429292d579ab",
      "samples_sha256": "6ee1411227a74b7813ed2113b120cfcab9870fa764d58074fc07e1faff6b8fb6",
      "model_wrapper": "internvl2",
      "model_args": "pretrained=OpenGVLab/InternVL2_5-1B,modality=video,device_map=auto,num_frame=8",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 991,
      "exactly_eight_parsed_labels": 31,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.082625,
        "grid_ade": 0.7740200084651281,
        "grid_fde": 1.2456937564708708,
        "grid_transition_acc": 0.025857142857142853,
        "token_f1": 0.16582469715689713
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "e10746ca190e2155eb7405b70b2bd89fc9d539e4b1eb13c63680ad5bcf547d86",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "segment_midpoint_mismatch",
      "input_sampling_policy": "eight_segment_midpoints",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": false,
      "reason": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required.",
      "artifact_integrity_pass": true,
      "model": "InternVL2.5 1B",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required."
    },
    {
      "method": "internvl2_5_4b",
      "display_name": "InternVL2.5 4B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.1145,
        "grid_ade": 0.5803833008996817,
        "grid_fde": 1.0010146711885595,
        "grid_transition_acc": 0.4007142857142857,
        "token_f1": 0.1740060606060606
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_110120_results.json",
      "result_sha256": "d70ba0f757f28a2dc034ac29b740f83bcb2f7e2af3890eee69fb09a0ea071b00",
      "samples_sha256": "3125a730aec246588bf98d45c4e00a84b9c2051b658ea8f4ca415424338bd4b3",
      "model_wrapper": "internvl2",
      "model_args": "pretrained=OpenGVLab/InternVL2_5-4B,modality=video,device_map=auto,num_frame=8",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 408,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.1145,
        "grid_ade": 0.5803833008996817,
        "grid_fde": 1.0010146711885595,
        "grid_transition_acc": 0.4007142857142857,
        "token_f1": 0.1740060606060606
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "a44ddc90105c3c0ea8bb9a9a5b576c5ddac23c32787dc315b18cddcd93470e16",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "segment_midpoint_mismatch",
      "input_sampling_policy": "eight_segment_midpoints",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": false,
      "reason": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required.",
      "artifact_integrity_pass": true,
      "model": "InternVL2.5 4B",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required."
    },
    {
      "method": "internvl2_5_8b",
      "display_name": "InternVL2.5 8B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.0475,
        "grid_ade": 0.6044177666316182,
        "grid_fde": 0.6155502631595502,
        "grid_transition_acc": 0.8097142857142858,
        "token_f1": 0.048625
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_124713_results.json",
      "result_sha256": "93c0969080f8b3718e9ece97c584fa3884c71590f4111f53ba01dea15890d5dc",
      "samples_sha256": "30da1feb8cf40329cf21be0441a3223c57c3bb44f2089228f1f439d782c2c18f",
      "model_wrapper": "internvl2",
      "model_args": "pretrained=OpenGVLab/InternVL2_5-8B,modality=video,device_map=auto,num_frame=8",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 1000,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.0475,
        "grid_ade": 0.6044177666316182,
        "grid_fde": 0.6155502631595502,
        "grid_transition_acc": 0.8097142857142858,
        "token_f1": 0.048625
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "9791f09897a550517dc531166d3bf81dfb7d974d85f632266230484a4a467662",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "segment_midpoint_mismatch",
      "input_sampling_policy": "eight_segment_midpoints",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": false,
      "reason": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required.",
      "artifact_integrity_pass": true,
      "model": "InternVL2.5 8B",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required."
    },
    {
      "method": "internvl3_1b",
      "display_name": "InternVL3 1B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.06075,
        "grid_ade": 0.9415124304496867,
        "grid_fde": 1.3895536924916783,
        "grid_transition_acc": 0.016428571428571428,
        "token_f1": 0.12500120957473898
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_131535_results.json",
      "result_sha256": "205863a69fb3ef80a72ee7eed54c52c300f149aadbd2ce4ee4ed76a1caeb9b31",
      "samples_sha256": "03ac9cc754a5dafb8e37fc1e87b6d56c0ba0bd76e5bf0421258bf3307062436f",
      "model_wrapper": "internvl2",
      "model_args": "pretrained=OpenGVLab/InternVL3-1B,modality=video,device_map=auto,num_frame=8",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 721,
      "exactly_eight_parsed_labels": 24,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.06075,
        "grid_ade": 0.9415124304496867,
        "grid_fde": 1.3895536924916783,
        "grid_transition_acc": 0.016428571428571428,
        "token_f1": 0.12500120957473898
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "0cf3363b6d343395ec21b179e4988e13bb323d26f6b0e8f547ab8a6ec4928f0c",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "segment_midpoint_mismatch",
      "input_sampling_policy": "eight_segment_midpoints",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": false,
      "reason": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required.",
      "artifact_integrity_pass": true,
      "model": "InternVL3 1B",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required."
    },
    {
      "method": "internvl3_2b",
      "display_name": "InternVL3 2B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.0515,
        "grid_ade": 0.5391038439465828,
        "grid_fde": 0.5193340442780277,
        "grid_transition_acc": 0.10557142857142855,
        "token_f1": 0.085
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_133652_results.json",
      "result_sha256": "5acef1386fe837d36c6accd7cb5e904d4f7210951bdc85a8d00ccce39e373a49",
      "samples_sha256": "62b696ccf4364b8f3bed6742505678789e521f15b3b5cc95a8e7d922bccdb3ab",
      "model_wrapper": "internvl2",
      "model_args": "pretrained=OpenGVLab/InternVL3-2B,modality=video,device_map=auto,num_frame=8",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 1000,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.0515,
        "grid_ade": 0.5391038439465828,
        "grid_fde": 0.5193340442780277,
        "grid_transition_acc": 0.10557142857142855,
        "token_f1": 0.085
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "bba62680acb6ae7eae79e35d6e50424394fd124c58dad7427d5b945c2c74b1bd",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "segment_midpoint_mismatch",
      "input_sampling_policy": "eight_segment_midpoints",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": false,
      "reason": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required.",
      "artifact_integrity_pass": true,
      "model": "InternVL3 2B",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required."
    },
    {
      "method": "internvl3_8b",
      "display_name": "InternVL3 8B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.176625,
        "grid_ade": 0.6052488269471503,
        "grid_fde": 0.6735395490582119,
        "grid_transition_acc": 0.5325714285714286,
        "token_f1": 0.19643055555555558
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_140222_results.json",
      "result_sha256": "a7de07fb80778725f757d700aae0f75767dcc4fbaeb9ce67ee2b09011462fcf6",
      "samples_sha256": "e12e5469a25b13b11e9e3131271965f800d48a43f69e7e96f06be3e857a9e315",
      "model_wrapper": "internvl2",
      "model_args": "pretrained=OpenGVLab/InternVL3-8B,modality=video,device_map=auto,num_frame=8",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 690,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.176625,
        "grid_ade": 0.6052488269471503,
        "grid_fde": 0.6735395490582119,
        "grid_transition_acc": 0.5325714285714286,
        "token_f1": 0.19643055555555558
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "66e1c00efa3ec66711c081315e7ae01dbe68ad03f17d944538f447f6d8bc22ab",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "segment_midpoint_mismatch",
      "input_sampling_policy": "eight_segment_midpoints",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": false,
      "reason": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required.",
      "artifact_integrity_pass": true,
      "model": "InternVL3 8B",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Wrapper samples segment midpoints but scorer reference uses endpoints; aligned-frame rerun required."
    },
    {
      "method": "videollama3_2b",
      "display_name": "VideoLLaMA3 2B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.104,
        "grid_ade": 0.539905614056961,
        "grid_fde": 0.7288593787161374,
        "grid_transition_acc": 0.016714285714285713,
        "token_f1": 0.18325028011204483
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_142215_results.json",
      "result_sha256": "c1d35fd381d0b8588430b03889cd742729c47b99cae3eb7a90d3fc530fa4c916",
      "samples_sha256": "873e1d1614a231b0317ca5a48ab19aaf970fdd5cf4e5fb8bc8c793323958b75e",
      "model_wrapper": "videollama3",
      "model_args": "pretrained=DAMO-NLP-SG/VideoLLaMA3-2B,device_map=auto,max_num_frames=8,use_custom_video_loader=True,use_flash_attention_2=False",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 595,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.104,
        "grid_ade": 0.539905614056961,
        "grid_fde": 0.7288593787161374,
        "grid_transition_acc": 0.016714285714285713,
        "token_f1": 0.18325028011204483
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "ae84e566f5d59290708c74f9a5c0706e1f37989e1c91fededaa90c15ba565caf",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "VideoLLaMA3 2B",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "videollama3_7b",
      "display_name": "VideoLLaMA3 7B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.022875,
        "grid_ade": 0.45601813877484937,
        "grid_fde": 0.4683676221702385,
        "grid_transition_acc": 0.5167142857142858,
        "token_f1": 0.02728475935828877
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_173446_results.json",
      "result_sha256": "1425afb6865494d9f0df2e2d67888201c52692cbdebdbd47542a4177ae03f850",
      "samples_sha256": "3fc1459e31501a82a328cfd15fb0b8b16daeb454fefe8923cea2063844aee25e",
      "model_wrapper": "videollama3",
      "model_args": "pretrained=DAMO-NLP-SG/VideoLLaMA3-7B,device_map=auto,max_num_frames=8,use_custom_video_loader=True,use_flash_attention_2=False",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 994,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.022875,
        "grid_ade": 0.45601813877484937,
        "grid_fde": 0.4683676221702385,
        "grid_transition_acc": 0.5167142857142858,
        "token_f1": 0.02728475935828877
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "1bf87c63d9e80eb528dda9f55095aac96a72204561dddb649186c39ebd013f68",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "VideoLLaMA3 7B",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "longva_7b",
      "display_name": "LongVA 7B",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.161625,
        "grid_ade": 0.4544350263422551,
        "grid_fde": 0.28540666145398175,
        "grid_transition_acc": 0.1307142857142857,
        "token_f1": 0.23936666666666667
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_180136_results.json",
      "result_sha256": "9679e95f75dcf8f3c4f45944064424df9219170f37e8640655372b1e2c864c92",
      "samples_sha256": "ae1d1c1049a97b0d75b41eec40ec5b8ddaff16f83ad6bcedbf9a91a5d3689a21",
      "model_wrapper": "longva",
      "model_args": "pretrained=lmms-lab/LongVA-7B,model_name=llava_qwen,conv_template=qwen_1_5,device_map=auto,max_frames_num=8,video_decode_backend=pyav",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 1000,
      "exactly_eight_parsed_labels": 940,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.161625,
        "grid_ade": 0.4544350263422551,
        "grid_fde": 0.28540666145398175,
        "grid_transition_acc": 0.1307142857142857,
        "token_f1": 0.23936666666666667
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "baf932a00fda70c114f5b890f8c6385a448ecd244e34d8466a77d8a045a3d801",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "LongVA 7B",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "phi4_multimodal",
      "display_name": "Phi-4 Multimodal Instruct",
      "samples": 1000,
      "source": "open",
      "metrics": {
        "grid_acc": 0.01125,
        "grid_ade": 1.05581984011649,
        "grid_fde": 1.080208712862938,
        "grid_transition_acc": 0.019714285714285715,
        "token_f1": 0.026647727272727274
      },
      "fresh_gpu_reproduction": false,
      "result_basename": "20260823_182402_results.json",
      "result_sha256": "9e8553db559e80254dbe24c0ac3cdfc2f2223c2bde84e675286261c2091cb1ed",
      "samples_sha256": "d91465a3ca94f4ab56ea6592712b82aea015fe1ed4314463e67e0dd8c1932052",
      "model_wrapper": "phi4_multimodal",
      "model_args": "pretrained=microsoft/Phi-4-multimodal-instruct,trust_remote_code=True,device_map=auto,dtype=bfloat16,max_frames_num=8,attn_implementation=eager",
      "recorded_git_hash": "7828431",
      "source_model_revision_pinned": false,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "ordered_first1000_identity_match": true,
      "unique_identity_count": 1000,
      "doc_id_order_match": true,
      "full_doc_content_match": true,
      "prompt_match": 1000,
      "target_match": 1000,
      "nonempty_predictions": 1000,
      "parsed_predictions": 379,
      "exactly_eight_parsed_labels": 342,
      "metric_recomputation_max_per_sample_abs_error": 0,
      "recomputed_metrics": {
        "grid_acc": 0.01125,
        "grid_ade": 1.05581984011649,
        "grid_fde": 1.080208712862938,
        "grid_transition_acc": 0.019714285714285715,
        "token_f1": 0.026647727272727274
      },
      "aggregate_metric_max_abs_error": 0,
      "log_sha256": "408dbe0f9e595f02d1e3019f0b5279c68a34679c947371488f84787e03d793ae",
      "fps_log_rows": 0,
      "logged_sampled_frame_counts": {},
      "logged_truncated_clips": 0,
      "error_log_markers": 0,
      "protocol_cohort": "endpoint_aligned_preview1000",
      "input_sampling_policy": "eight_endpoint_inclusive_uniform",
      "reference_sampling_policy": "eight_endpoint_inclusive_uniform",
      "rank_eligible": true,
      "reason": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun.",
      "artifact_integrity_pass": true,
      "model": "Phi-4 Multimodal Instruct",
      "effective_fps": null,
      "protocol_status": "aligned_preview",
      "score_record_prompt_match": 1000,
      "prompt_evidence_field": "input (arguments stores field names, not the prompt)",
      "protocol_note": "Archived 1000/1000 ordered identities, original annotations, prompts, targets and all five per-sample/aggregate metrics verified. Frame-sampling policy is compatible according to saved configuration and clean tracked wrapper source; per-frame decoder hashes and exact model revisions are not archived. Not a fresh GPU rerun."
    },
    {
      "method": "gemini-3.1-pro-preview",
      "display_name": "Gemini 3.1 Pro Preview (gemini-3.1-pro-preview)",
      "samples": 1000,
      "source": "gemini",
      "metrics": {
        "grid_acc": 0.06601897790489211,
        "grid_ade": 0.7798144629669007,
        "grid_fde": 0.8496851803840997,
        "grid_transition_acc": 0.603178248289792,
        "token_f1": 0
      },
      "fresh_gpu_reproduction": false,
      "samples_sha256": "9acadc39debda1f150e16d1c61448be39f556ae9f9c9843aac199f35069b08bc",
      "summary_sha256": "33e13bc42e2acf3d2f305cd7538c084d623b0dc705d65826f07c298b947af752",
      "protocol_cohort": "api_legacy_full_trajectory",
      "input_sampling_policy": "target4fps_max8_frames_4to8_actual",
      "reference_sampling_policy": "full_original_trajectory",
      "rank_eligible": false,
      "reason": "Historical API protocol: full-trajectory targets, JSON-answer prompt, max 1024 output tokens and 4–8 sampled input frames at target 4 FPS. Different scorer text normalization; not comparable with canonical eight-frame endpoint-aligned task. All 1000 calls returned; malformed/empty parsed answers were retained in reported means.",
      "sample_count": 1000,
      "successful_calls": 1000,
      "nonempty_predictions": 998,
      "actual_sampled_frame_counts": {
        "4": 5,
        "5": 14,
        "6": 31,
        "7": 49,
        "8": 901
      },
      "ordered_first1000_video_match": true,
      "original_full_target_match": 1000,
      "canonical_target_match": 0,
      "canonical_prompt_match": 0,
      "recomputed_stored_means": {
        "grid_acc": 0.06601897790489213,
        "grid_ade": 0.7798144629669005,
        "grid_fde": 0.849685180384094,
        "grid_transition_acc": 0.6031782482897919,
        "token_f1": 0
      },
      "aggregate_metric_max_abs_error": 5.662137425588298e-15,
      "artifact_integrity_pass": true,
      "model": "Gemini 3.1 Pro Preview (gemini-3.1-pro-preview)",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "valid_parsed_grid_predictions": 952,
      "nonempty_parsed_predictions": 998,
      "parse_error_records": 2,
      "api_transport_failures": 0,
      "grid_metric_recomputation_max_abs_error": 9.170442183403793e-14,
      "grid_metric_recomputation_prediction_field": "prediction_parsed",
      "api_grid_scorer": "scripts/api_baselines/dive_metrics.py::compute_grid_metrics",
      "ordered_first1000_qid_match": true,
      "ordered_first1000_identity_match": true,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "protocol_note": "Historical API protocol: full-trajectory targets, JSON-answer prompt, max 1024 output tokens and 4–8 sampled input frames at target 4 FPS. Different scorer text normalization; not comparable with canonical eight-frame endpoint-aligned task. All 1000 calls returned; malformed/empty parsed answers were retained in reported means."
    },
    {
      "method": "gemini-3.6-flash",
      "display_name": "Gemini 3.6 Flash (gemini-3.6-flash)",
      "samples": 1000,
      "source": "gemini",
      "metrics": {
        "grid_acc": 0.08850071741772678,
        "grid_ade": 0.5026989594846982,
        "grid_fde": 0.5715094860334752,
        "grid_transition_acc": 0.8637878798555269,
        "token_f1": 0
      },
      "fresh_gpu_reproduction": false,
      "samples_sha256": "f788f55a70a1860ccf9d58c5a5f448e171bd78c447fda1160f0460dfa235a727",
      "summary_sha256": "4d3545f53a27c8c7a692124681e8200e1928de67afa94a797e66924ff46616b2",
      "protocol_cohort": "api_legacy_full_trajectory",
      "input_sampling_policy": "target4fps_max8_frames_4to8_actual",
      "reference_sampling_policy": "full_original_trajectory",
      "rank_eligible": false,
      "reason": "Historical API protocol: full-trajectory targets, JSON-answer prompt, max 1024 output tokens and 4–8 sampled input frames at target 4 FPS. Different scorer text normalization; not comparable with canonical eight-frame endpoint-aligned task. All 1000 calls returned; malformed/empty parsed answers were retained in reported means.",
      "sample_count": 1000,
      "successful_calls": 1000,
      "nonempty_predictions": 1000,
      "actual_sampled_frame_counts": {
        "4": 5,
        "5": 14,
        "6": 31,
        "7": 49,
        "8": 901
      },
      "ordered_first1000_video_match": true,
      "original_full_target_match": 1000,
      "canonical_target_match": 0,
      "canonical_prompt_match": 0,
      "recomputed_stored_means": {
        "grid_acc": 0.0885007174177267,
        "grid_ade": 0.5026989594846988,
        "grid_fde": 0.5715094860334722,
        "grid_transition_acc": 0.8637878798555271,
        "token_f1": 0
      },
      "aggregate_metric_max_abs_error": 2.9976021664879227e-15,
      "artifact_integrity_pass": true,
      "model": "Gemini 3.6 Flash (gemini-3.6-flash)",
      "effective_fps": null,
      "protocol_status": "historical_unaligned",
      "valid_parsed_grid_predictions": 896,
      "nonempty_parsed_predictions": 1000,
      "parse_error_records": 0,
      "api_transport_failures": 0,
      "grid_metric_recomputation_max_abs_error": 9.103828801926284e-14,
      "grid_metric_recomputation_prediction_field": "prediction_parsed",
      "api_grid_scorer": "scripts/api_baselines/dive_metrics.py::compute_grid_metrics",
      "ordered_first1000_qid_match": true,
      "ordered_first1000_identity_match": true,
      "ordered_identity_sha256": "e0ee94d90db19c464e5e87a12f1e90cfb6b5fc1e0e7276cf03e4c1e885c40f0e",
      "protocol_note": "Historical API protocol: full-trajectory targets, JSON-answer prompt, max 1024 output tokens and 4–8 sampled input frames at target 4 FPS. Different scorer text normalization; not comparable with canonical eight-frame endpoint-aligned task. All 1000 calls returned; malformed/empty parsed answers were retained in reported means."
    }
  ],
  "audit_method_notes": [
    "Read-only verification against local archived run/result/sample files, original pinned Parquet and released scorer; no inference, API generation or source edits.",
    "Pure deterministic reference-sequence memoization avoided re-parsing identical full trajectories for every model. Prompt verification used archived input plus score-record question, with the validated eight-label target length.",
    "All 27 archived rows may be displayed transparently, but only 18 belong in the protocol-compatible preview ranking. Six InternVL midpoint runs, one variable-frame/clipped GRT run and two full-trajectory API runs must be unranked historical records.",
    "No model_revision argument was pinned in these 25 archived open-model configurations. Clean current tracked wrapper source and saved configurations establish intended policy, not a fresh hardware or exact frame-byte replay."
  ],
  "source_manifest_checks": {
    "leaderboard_csv": true,
    "leaderboard_markdown": true,
    "api_summary_hashes": 4,
    "new_summary_csv": true,
    "obsolete_educational_background_summary_path_available": false
  },
  "cohort_counts": {
    "aligned_preview": 18,
    "historical_unaligned": 9
  },
  "grt_vs_llava_hf_0_5b": {
    "matched_ablation": false,
    "quality_win": false,
    "reason": "Different wrapper/checkpoint conversion and input protocol; recorded GRT quality is lower on all five metrics, but this is not a matched algorithm ablation.",
    "grt_metrics": {
      "grid_acc": 0.10125,
      "grid_ade": 0.712142440040729,
      "grid_fde": 0.6864884792587348,
      "grid_transition_acc": 0.014857142857142855,
      "token_f1": 0.18128055327025913
    },
    "baseline_metrics": {
      "grid_acc": 0.142875,
      "grid_ade": 0.5091069196938063,
      "grid_fde": 0.46643800755695713,
      "grid_transition_acc": 0.01714285714285714,
      "token_f1": 0.21888529905064952
    },
    "grt_minus_baseline": {
      "grid_acc": -0.041624999999999995,
      "grid_ade": 0.20303552034692274,
      "grid_fde": 0.22005047170177766,
      "grid_transition_acc": -0.002285714285714285,
      "token_f1": -0.037604745780390386
    }
  }
}
