{
  "as_of": "2026-10-11",
  "errors": [],
  "checks": [
    "Required fields",
    "Unique record IDs",
    "Source foreign keys",
    "Finite numbers",
    "ISO report dates",
    "Nonnegative integer counts",
    "GPU topology consistency",
    "Metric units",
    "Exact record deduplication",
    "Conflicts at identical source locators",
    "Source revision review",
    "Input SHA-256 fingerprints",
    "SQLite integrity and foreign-key checks",
    "Model repository mapping conflicts and evidence fields",
    "Source-run grouping and foreign keys",
    "Original source URL on every measurement"
  ],
  "record_count": 9917,
  "source_count": 108,
  "flagged_records": 9340,
  "exact_duplicates_removed": 45,
  "compilation_duplicates_removed": 0,
  "upstream_duplicate_locators": [
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [0]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [10]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [11]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [1]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [2]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [3]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [4]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [5]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [6]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [7]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [8]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1029-buildcmp-stew675/stock-vulkan/llama-bench.json [9]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [0]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [10]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [1]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [2]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [3]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [4]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [5]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [6]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [7]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [8]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1035-model-buildcmp-stock-rocm/llama-bench.json [9]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [0]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [10]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [1]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [2]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [3]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [4]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [5]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [6]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [7]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [8]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1040-model-buildcmp-stew675-vulkan/llama-bench.json [9]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [0]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [10]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [1]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [2]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [3]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [4]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [5]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [6]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [7]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [8]",
    "dp-craft-llamabench :: bench/runs/2026-09-15-1046-model-buildcmp-stew675-rocm/llama-bench.json [9]"
  ],
  "duplicate_record_ids": [],
  "conflicting_locator_groups": 0,
  "flags": {
    "accuracy_regression_unresolved": {
      "count": 20,
      "title": "Accuracy regression unresolved",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "actual_token_lengths_missing": {
      "count": 164,
      "title": "Actual token lengths missing",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "agentic_uncontrolled_trajectory": {
      "count": 180,
      "title": "Agentic uncontrolled trajectory",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "agentic_uncontrolled_workload": {
      "count": 2541,
      "title": "Agentic uncontrolled workload",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "aggregate_input_output_throughput": {
      "count": 4,
      "title": "Aggregate input output throughput",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "approximate_phase_rates": {
      "count": 5,
      "title": "Approximate phase rates",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "approximate_report": {
      "count": 2,
      "title": "Approximate report",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "attempt_budget_unreported": {
      "count": 15,
      "title": "Attempt budget unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "author_questioned_initial_result": {
      "count": 5,
      "title": "Author questioned initial result",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "backend_mismatch": {
      "count": 5,
      "title": "Backend mismatch",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "backend_unverified": {
      "count": 28,
      "title": "Backend unverified",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "board_chip_count_ambiguous": {
      "count": 12,
      "title": "Board chip count ambiguous",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "cache_reuse_likely": {
      "count": 3,
      "title": "Cache reuse likely",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "checkpoint_unreported": {
      "count": 29,
      "title": "Checkpoint unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "cold_run_thermal_comparison_caveat": {
      "count": 20,
      "title": "Cold run thermal comparison caveat",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "community_hardware_report": {
      "count": 15,
      "title": "Community hardware report",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "community_submission": {
      "count": 28,
      "title": "Community submission",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "companion_article_conflict": {
      "count": 16,
      "title": "Companion article conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "conflicting_system_configuration": {
      "count": 9,
      "title": "Conflicting system configuration",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "context_approximate": {
      "count": 1,
      "title": "Context approximate",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "corpus_hash_conflict": {
      "count": 12,
      "title": "Corpus hash conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "corrected_result": {
      "count": 18,
      "title": "Corrected result",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "correction_to_prior_claim": {
      "count": 13,
      "title": "Correction to prior claim",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "cpu_offload": {
      "count": 31,
      "title": "Cpu offload",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "custom_backend": {
      "count": 204,
      "title": "Custom backend",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "custom_quality_probe": {
      "count": 24,
      "title": "Custom quality probe",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "dataset_revision_unreported": {
      "count": 34,
      "title": "Dataset revision unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "deprecated_engine": {
      "count": 2,
      "title": "Deprecated engine",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "derived_from_recorded_booleans": {
      "count": 21,
      "title": "Derived from recorded booleans",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "derived_test_score": {
      "count": 5,
      "title": "Derived test score",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "engine_method_ambiguity": {
      "count": 2,
      "title": "Engine method ambiguity",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "engine_version_missing": {
      "count": 1,
      "title": "Engine version missing",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "evaluation_subset": {
      "count": 54,
      "title": "Evaluation subset",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "evaluation_timeout_failure": {
      "count": 2,
      "title": "Evaluation timeout failure",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "exact_checkpoint_unknown": {
      "count": 49,
      "title": "Exact checkpoint unknown",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "exact_checkpoint_unresolved": {
      "count": 472,
      "title": "Exact checkpoint unresolved",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "exact_run_settings_incomplete": {
      "count": 8,
      "title": "Exact run settings incomplete",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "experimental_build": {
      "count": 94,
      "title": "Experimental build",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "experimental_speculative_path": {
      "count": 840,
      "title": "Experimental speculative path",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "failed_predeclared_effect_threshold": {
      "count": 1,
      "title": "Failed predeclared effect threshold",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "feature_branch_later_clarified": {
      "count": 8,
      "title": "Feature branch later clarified",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "first_visible_content_latency": {
      "count": 143,
      "title": "First visible content latency",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "followup_campaign": {
      "count": 81,
      "title": "Followup campaign",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "gpu_assignment_ambiguous": {
      "count": 534,
      "title": "GPU assignment ambiguous",
      "description": "The raw log enumerates multiple devices without proving which participated in the run."
    },
    "gpu_parallelism_unreported": {
      "count": 20,
      "title": "Gpu parallelism unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "gpu_participation_unreported": {
      "count": 28,
      "title": "Gpu participation unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "greedy_nondeterminism_reported": {
      "count": 20,
      "title": "Greedy nondeterminism reported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "hardware_provided_by_vendor": {
      "count": 35,
      "title": "Hardware provided by vendor",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "hardware_specification_conflict": {
      "count": 446,
      "title": "Hardware specification conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "headline_statistic_unlabeled": {
      "count": 1,
      "title": "Headline statistic unlabeled",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "historical_implementation": {
      "count": 18,
      "title": "Historical implementation",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "historical_profile": {
      "count": 16,
      "title": "Historical profile",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "historical_reverted_configuration": {
      "count": 50,
      "title": "Historical reverted configuration",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "historical_weight_revision_unreported": {
      "count": 54,
      "title": "Historical weight revision unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "host_configuration_missing": {
      "count": 16,
      "title": "Host configuration missing",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "informal_peak": {
      "count": 3,
      "title": "Informal peak",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "informal_report": {
      "count": 7,
      "title": "Informal report",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "instance_semantics_unspecified": {
      "count": 10,
      "title": "Instance semantics unspecified",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "large_scenario_outlier": {
      "count": 4,
      "title": "Large scenario outlier",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "later_model_recommendation_changed": {
      "count": 20,
      "title": "Later model recommendation changed",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "lora_not_full_finetune": {
      "count": 8,
      "title": "Lora not full finetune",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "mean_median_comparison_conflict": {
      "count": 1,
      "title": "Mean median comparison conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "memory_scope_varies": {
      "count": 2,
      "title": "Memory scope varies",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "memory_variant_unreported": {
      "count": 2,
      "title": "Memory variant unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "missing_concurrency": {
      "count": 4009,
      "title": "Concurrency not reported",
      "description": "The number of active sequences or requests is not explicitly recorded."
    },
    "missing_context": {
      "count": 4251,
      "title": "Context capacity not reported",
      "description": "The configured context capacity is not reported. Prompt length or pre-existing KV depth, when known, is retained separately."
    },
    "missing_date": {
      "count": 5124,
      "title": "Exact report date unknown",
      "description": "A calendar date for this measurement is unavailable. Relative dates and archive names are not silently promoted to exact measurement dates."
    },
    "missing_gpu_count": {
      "count": 586,
      "title": "GPU participation unknown",
      "description": "The report does not establish how many GPU chips participated. Enumerating devices or counting physical boards is not allocation evidence."
    },
    "missing_os": {
      "count": 2221,
      "title": "OS not reported",
      "description": "The operating system for this run is unavailable; a repository setup guide is not automatically evidence for every archived campaign."
    },
    "missing_parallelism": {
      "count": 875,
      "title": "Multi-GPU split unknown",
      "description": "Multiple GPUs are reported, but the participating layout or parallelization method is not established."
    },
    "missing_quantization": {
      "count": 296,
      "title": "Precision not reported",
      "description": "The source does not establish this model or workload precision."
    },
    "missing_scope": {
      "count": 112,
      "title": "Measurement scope unclear",
      "description": "Per-request, aggregate, or per-device scope is not adequately documented."
    },
    "missing_test_context": {
      "count": 21,
      "title": "Missing test context",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "mixed_configuration_capture": {
      "count": 4,
      "title": "Mixed configuration capture",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "mixed_gpu_inventory": {
      "count": 12,
      "title": "Mixed gpu inventory",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "mixed_gpu_system": {
      "count": 4,
      "title": "Mixed gpu system",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "model_developer_report": {
      "count": 4,
      "title": "Model developer report",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "model_from_thread_context": {
      "count": 3,
      "title": "Model from thread context",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "model_identifier_sanitized": {
      "count": 12,
      "title": "Model identifier sanitized",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "model_identity_incomplete": {
      "count": 2,
      "title": "Model identity incomplete",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "model_label_conflict": {
      "count": 34,
      "title": "Model labels disagree",
      "description": "The section, filename, or table gives conflicting model identification. Original labels are retained in the settings."
    },
    "model_sanity_check_not_perfect": {
      "count": 60,
      "title": "Model sanity check not perfect",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "model_variant_missing": {
      "count": 12,
      "title": "Model variant missing",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "narrative_split_mode_unsupported": {
      "count": 8,
      "title": "Narrative split mode unsupported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "no_raw_timing_log": {
      "count": 5,
      "title": "No raw timing log",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "nonmonotonic_prefill": {
      "count": 1,
      "title": "Nonmonotonic prefill",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "nonstandard_unit": {
      "count": 10,
      "title": "Unexpected metric unit",
      "description": "The reported unit differs from the common unit for this metric. It has not been silently converted."
    },
    "not_official_swe_bench_resolved": {
      "count": 5,
      "title": "Not official swe bench resolved",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "not_raw_throughput": {
      "count": 9,
      "title": "Not raw throughput",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "os_metadata_static": {
      "count": 550,
      "title": "Os metadata static",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "outlier_exclusion": {
      "count": 5,
      "title": "Outlier exclusion",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "packaged_image_not_full_benchmarked": {
      "count": 38,
      "title": "Packaged image not full benchmarked",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "parallelism_unreported": {
      "count": 3,
      "title": "Parallelism unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "partial_configuration": {
      "count": 30,
      "title": "Partial configuration",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "partial_failed_run": {
      "count": 4,
      "title": "Partial failed run",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "possible_dtype_weight_conflation": {
      "count": 3,
      "title": "Possible dtype weight conflation",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "power_monitor_scope_ambiguous": {
      "count": 4,
      "title": "Power monitor scope ambiguous",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "power_scope_unreported": {
      "count": 13,
      "title": "Power scope unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "precision_fit_inconsistency": {
      "count": 3,
      "title": "Precision fit inconsistency",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "preliminary": {
      "count": 2,
      "title": "Preliminary",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "prompt_format_mismatch": {
      "count": 63,
      "title": "Prompt format mismatch",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "quality_affected_by_token_cap": {
      "count": 7,
      "title": "Quality affected by token cap",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "quantization_bit_width_missing": {
      "count": 1,
      "title": "Quantization bit width missing",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "quantization_campaign_inherited": {
      "count": 400,
      "title": "Quantization campaign inherited",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "quantization_label_conflict": {
      "count": 20,
      "title": "Quantization label conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "quantization_missing": {
      "count": 122,
      "title": "Quantization missing",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "queueing_latency": {
      "count": 9,
      "title": "Queueing latency",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "reasoning_time_may_be_included": {
      "count": 2,
      "title": "Reasoning time may be included",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "relative_metric": {
      "count": 14,
      "title": "Relative metric",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "repeat_run_kept_separate": {
      "count": 6,
      "title": "Repeat run kept separate",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "reported_energy_ratio_mismatch": {
      "count": 1,
      "title": "Reported energy ratio mismatch",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "reported_tokens_exceed_configured_cap": {
      "count": 28,
      "title": "Reported tokens exceed configured cap",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "request_shape_missing": {
      "count": 136,
      "title": "Request shape missing",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "retained_performance_state_outlier": {
      "count": 5,
      "title": "Retained performance state outlier",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "rounded_context_label": {
      "count": 56,
      "title": "Rounded context label",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "run_inference_profile_unreported": {
      "count": 15,
      "title": "Run inference profile unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "saturated_probe_not_ranking": {
      "count": 24,
      "title": "Saturated probe not ranking",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "selected_models_unreported": {
      "count": 6,
      "title": "Selected models unreported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "self_reported": {
      "count": 67,
      "title": "Self reported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "served_alias_only": {
      "count": 18,
      "title": "Served alias only",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "short_prompt_only": {
      "count": 8,
      "title": "Short prompt only",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "single_observation_per_category": {
      "count": 32,
      "title": "Single observation per category",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "single_profile_not_three_seed_ladder": {
      "count": 8,
      "title": "Single profile not three seed ladder",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "single_repetition": {
      "count": 541,
      "title": "Single repetition",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "small_sample": {
      "count": 16,
      "title": "Small sample",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_baseline_conflict": {
      "count": 18,
      "title": "Source baseline conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_configuration_conflict": {
      "count": 50,
      "title": "Source configuration conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_denominator_corrected_from_raw": {
      "count": 5,
      "title": "Source denominator corrected from raw",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_hardware_conflict": {
      "count": 47,
      "title": "Source hardware conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_hardware_description_conflict": {
      "count": 5,
      "title": "Source hardware description conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_heading_backend_conflict": {
      "count": 6,
      "title": "Source heading backend conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_internal_conflict": {
      "count": 6,
      "title": "Source details conflict",
      "description": "The source contains contradictory configuration or measurement statements. The inconsistency is preserved."
    },
    "source_methodology_conflict": {
      "count": 15,
      "title": "Source methodology conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_model_name_conflict": {
      "count": 1,
      "title": "Source model name conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_platform_conflict": {
      "count": 207,
      "title": "Source platform conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_quantization_summary_conflict": {
      "count": 1,
      "title": "Source quantization summary conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_relative_claim_mismatch": {
      "count": 4,
      "title": "Source relative claim mismatch",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_summary_incomplete": {
      "count": 6,
      "title": "Source summary incomplete",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "source_vendor": {
      "count": 73,
      "title": "Vendor-reported result",
      "description": "The organization supplying the hardware or software reports this measurement; documentation strength is separate from independence."
    },
    "source_version_label_conflict": {
      "count": 8,
      "title": "Source version label conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "speculative_decoding": {
      "count": 2,
      "title": "Speculative decoding",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "sponsored": {
      "count": 30,
      "title": "Sponsored",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "stale_prose_conflict": {
      "count": 18,
      "title": "Stale prose conflict",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "stream_chunk_token_interpolation": {
      "count": 288,
      "title": "Stream chunk token interpolation",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "tail_estimate_small_sample": {
      "count": 408,
      "title": "Tail estimate small sample",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "test_collection_errors": {
      "count": 15,
      "title": "Test collection errors",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "thermal_throttling_reported": {
      "count": 32,
      "title": "Thermal throttling reported",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "throughput_scope_unspecified": {
      "count": 40,
      "title": "Throughput scope unspecified",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "token_accounting_inconsistency": {
      "count": 4,
      "title": "Token accounting inconsistency",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "tool_qualification_failed": {
      "count": 84,
      "title": "Tool qualification failed",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "ttft_definition_not_explicit": {
      "count": 2,
      "title": "Ttft definition not explicit",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "ttft_includes_reasoning_possible": {
      "count": 2,
      "title": "Ttft includes reasoning possible",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "tuned_power_profile": {
      "count": 204,
      "title": "Tuned power profile",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "uncontrolled_configuration_changes": {
      "count": 6,
      "title": "Uncontrolled configuration changes",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "unknown_quantization": {
      "count": 53,
      "title": "Unknown quantization",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "variable_output_length_confound": {
      "count": 180,
      "title": "Variable output length confound",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "vendor_claim": {
      "count": 39,
      "title": "Vendor claim",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "warmup_disabled": {
      "count": 6,
      "title": "Warmup disabled",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "work_normalized_metric": {
      "count": 1,
      "title": "Work normalized metric",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    },
    "zero_reported_variance": {
      "count": 1398,
      "title": "Zero reported variance",
      "description": "Source-specific caveat. Open an affected record to inspect its notes, exact settings, and source locator."
    }
  },
  "input_sha256": {
    "cards.json": "30133eb6ed55b18542a704b68c7a6e301d467f4f2e5854ab50a5d23f5ed7a81a",
    "community.json": "bab32e6512b46dbb55af3e64ee8f71f4db452273ac14db30d400ea9fbb4a985e",
    "evaluation-agent-scores.json": "ff687c645c0ced700aa2619c24487d648d1de9e79a561501509c567f4d467451",
    "evaluation-pi-bench.json": "68c5ad64f5ca5dbca9ab1556b376cd4b2a5036c34a264323e3094d37fb922d74",
    "expansion-labs.json": "45aa7fe502b7fe7d4385230012c9cfb4b8a2fbe51f19d29b93b9f21c3a7fd3e4",
    "expansion-owner.json": "16abb198c74720aed54f97463d42281b44dc136cc59ebffaa1b41edc0e06079f",
    "expansion-rare.json": "7b86cab12fdb557d7c9bdf160adb4df9f9d3e5a10aa6217f989da23b0c4005e9",
    "expansion-raw.json": "768b0a6a6b5a2ee733befa77a382d3fdff45115fa333da1291fc9977a7452bdf",
    "expansion-vendor.json": "d9ba985dd7b091a2ea3eb1d64e2d21f2e72da2fe70a7cc53dcb2e2bc7c029e1d",
    "labs.json": "f0c89aae311cac2a9d13b4fde1c0db4d04091b1681e8993e708a60b889d46f2a",
    "model-links-evaluation-agent-scores.json": "ba4e07c56f1c36756157ac5af4ce8d7c804af7a897d7880bebde9127dada0d9c",
    "model-links-evaluation-pi-bench.json": "e8367ade1c35d3ef6e9aaeee9ac805b0dcf994767cb368d407ea3719c0b1ad21",
    "model-links-labs.json": "f8af915b48cc1817fb65743e1592c0915697938ceb9cba4777e41114e40ebc35",
    "model-links-raw.json": "dcb1c1a3a0b67e3d292f780f41c56befd7d4b227ec6266ca5b4cdec2c40621fb",
    "model-links-vendor.json": "439fafed815ebaf01bc3c84d1268411a42b67d8212e76f0702824def67709b80",
    "raw.json": "f6d37bda25fadb66e239c82e97f6e7f4789783a163620f30e5035b4743c20ab5",
    "vendor.json": "e2f294a4578083bcb9cf47b411e35cad7d2d2aa3854234c47827365bacb2e773",
    "evaluation-agent-scores-quality.md": "4cc06e31c7b02cebdd7525f67e0f4892d2ce8a7a6ce91bba234237b6e9a2d5e5",
    "evaluation-pi-bench-quality.md": "cbdee29cedef17af53e4cfe0bddd1e9a9dfd10989f7d0398be1b961cd7932112",
    "evaluation-source-navigation-quality.md": "e7bff585ed366d428335eaa3b3f43b2de9c8e3a3a587fd63f0f3e6864efbe804",
    "expansion-labs-quality.md": "9467c6d8b38fa8032b7977bd41d1dc27fe108f8f543b6db4436c0c066f99cc3e",
    "expansion-owner-quality.md": "67c737e752c06992aa558083cafee5ba6186a037b0d737798963c34844558243",
    "expansion-rare-quality.md": "61a5c0f0046ff5ccb43a43be96b51c9f786761b25fcb2a372ae8a503b25c22b0",
    "expansion-raw-quality.md": "d2c084d8a5fc8430884a1a40eb11f0189318d63be927a2b9de2ab9746c2c9904",
    "expansion-vendor-quality.md": "156e47718e39948e1e06647dce854515916531df48675002756f93b83ed80f0c",
    "llm-scope-quality.md": "11da71a7a63171412cadefba0d4f38b63f7e59dcc1ae51c31471d9c04527f72a",
    "model-links-labs-quality.md": "6f5561b2e8920c43fd5bb3738c40682d9fff3b78af2626bdb86845c2fc06ae62",
    "model-links-raw-quality.md": "9789cf98d41d42dfcd8cc4ae10b15eef0754007dce6c1a9559b0d62edbb4f9a7",
    "model-links-vendor-quality.md": "7cacaa768ed55b267a994c6d7fce01a07a5dbbf2586f91311e0142b07d084182"
  },
  "review_notes": [
    {
      "title": "Evaluation Agent Scores Quality",
      "text": "# Evaluation source and second-pass audit\n\nReviewed 2026-10-11. This bounded lane reads original data and harness source; no external code, model, evaluator or test suite was executed. Only the three assigned research files were written.\n\n## Deliverables\n\n- `evaluation-agent-scores.json`: 4 campaign sources, 78 score rows and 12 coverage entries.\n- `model-links-evaluation-agent-scores.json`: 10 source/model mappings, 9 identifying source-named tested checkpoints and 1 upstream-only reference.\n- Row levels: 17 overall results and 61 explicitly labeled subtests. Categories, individual terminal tasks and prompt-depth rungs use `evaluation_level: subtest`; aggregate scores use `overall`.\n- Scores preserve fractions or task rewards from the source. Numerator, denominator, scale, subset, scoring method and evaluation status are explicit. Timing is metadata, not a quality score.\n\n## Pinned evidence\n\n[mattbucci repository](https://github.com/mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference/tree/9d43d97db04df090d76bad8ff813a96c3189c198) is pinned at `9d43d97db04df090d76bad8ff813a96c3189c198` (2026-10-09). The six April model configurations are instead bound to the [original quality-publication commit](https://github.com/mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference/tree/ae3ca9c1c23f74f611dd59b22a694f0f0c442563), preventing newer launcher defaults from changing their identity.\n\n[Terminal-Bench-Local](https://github.com/kyuz0/terminal-bench-mini/tree/07034484346dc724d0e2c47c821fd196add1d6fb) is pinned at `07034484346dc724d0e2c47c821fd196add1d6fb` (2026-09-27). Only its eligible dual-R9700 result was imported; Strix/Gorgon Halo systems are outside the hardware catalog.\n\n## Retained results and method limits\n\nMMLU contributes 5 scores, HumanEval 5 and LAB-Bench 24 (3 overall plus 21 category scores). MMLU actually contains 57 questions, one first item from each subject, despite the prose/CLI request saying 100. HumanEval is the first 30 tasks. LAB-Bench samples 25 items from each of 7 train-split categories. These are custom subsets, not official full-dataset leaderboard scores; dataset revisions are unreported.\n\nThe historical harness uses a 512-token multiple-choice budget and up to 4096 code tokens. The later Qwen3.8 harness uses 1024 MC tokens. Both the old code and publication footnotes were read; the current 1024 default was not applied to April. LAB-Bench answer-option ordering uses Python's process-dependent hash without a recorded PYTHONHASHSEED, limiting exact replay.\n\nHistorical presets resolve exact local checkpoint names. Qwen3.5-MoE is official GPTQ Int4 for experts with other weights BF16, despite the commit message's blanket AWQ heading. The old README explicitly identifies EndeavourOS, Ryzen 9 7900 and 64 GB RAM. Its ROCm 7.2.0/7.2.1 conflict is preserved; its GDDR7 typo was not copied into a memory-type field. No old host configuration was inherited by later campaigns without evidence.\n\nThe post095 tool-use ladders score 21/21 final seed-rungs for each of North and Laguna. Prompt hashes match across the three effective seeds. Qwen3.8 scores 7/7 under a different, single greedy repeated-filler profile. All primary/follow-up outcomes, budgets, calibration attempts and actual prompt lengths were checked. These saturated synthetic probes cannot rank general coding competence. Their 7 depth subtests are separate from each overall result.\n\nTerminal-Bench-Local Core19 v1.0.0 scores 17/19 under Harbor 0.20.0 and Terminus-2 v2.0.0, with up to two attempts and stop-on-pass. All 19 task files and 21 attempts were inspected. COBOL modernization and the inference-batching scheduler each timed out twice at 10800 seconds; both failures remain in the denominator. This campaign exhausted its attempt policy, even though those task records have `completed:false`. Recomputed duration 105408549 ms matches the summary.\n\nThe Terminal result's explicit `dual-r9700` platform label establishes two R9700s. It does not establish tensor/layer split, CPU offload, host OS/CPU/RAM or per-GPU memory. Those fields stay unknown. Its exact local GGUF filename is preserved, but no quantizer namespace/hash is reported; the model link is upstream-only even though an identically named Unsloth file exists.\n\n## Exclusions and unresolved coverage\n\nThe [SWE-bench setup](https://github.com/mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference/blob/9d43d97db04df090d76bad8ff813a96c3189c198/evals/swebench/FP8_BAKEOFF_SETUP.md) and README classify v2 as an answer-exposure study and acknowledge equivalent leak channels in July. No scores from either are imported. v4/v5 restarts and rollout-completion counts are not solved-task counts.\n\nThe [October 3 v5 audit](https://github.com/mattbucci/2x-R9700-RDNA4-GFX1201-sglang-inference/blob/9d43d97db04df090d76bad8ff813a96c3189c198/evals/swebench/git-peek-audit-v5-2026-10-03.json) still marks 14/300 opencode and 9/244 opencode-dcp instances exposed. The pinned tree supplies no complete clean resolved-score matrix. Its future zero-exposure gate is an intended requirement, not proof that existing runs met it.\n\nForty-four problematic canonical score cells were excluded: 40 LAB-Bench category/overall cells from five suspect thinking-budget tables, 2 false-low Qwen3.5 MMLU cells, and 2 Gemma HumanEval cells whose endpoint compatibility was explicitly questioned. Retained Qwen3.8 LAB-Bench is the corrected no-think result. Pre-fix North ladder failures are not reused as current quality evidence.\n\nTargeted DeepSWE searches covered R9700, W7800, W7900 and W6800, GitHub, Reddit and the [official leaderboard](https://deepswe.datacurve.ai/). No verified eligible-PRO run with a defensible score denominator was found. MoziAI cards combine hardware deployment advice with broad benchmark claims but do not attribute those scores to a specific R9700 evaluation; they are recorded as exclusions. This is a bounded coverage finding, not a claim that no such result exists anywhere.\n\n## Second quality pass\n\nRe-read historical configuration and scoring code after the initial extraction; this caught the 512/1024 budget change, two additional endpoint caveats and the GPTQ/AWQ distinction. Rechecked the later leak-audit counters against the intended isolation description. Reconciled every retained LAB-Bench total, terminal task/attempt count and ladder final outcome; no numerator/rate discrepancy remains.\n\nFresh second downloads of the pinned Qwen3.8 summary, North ladder receipt and Terminal summary matched their first reads exactly. All 78 rows have valid source and model mappings, finite bounded scores and exact numerator/denominator agreement. Model metadata was checked through public Hugging Face APIs; three historical mattbucci URLs now redirect to renamed repositories and carry explicit revision caveats. No current model SHA is represented as a historical tested revision.\n\n\nRevision status remains unknown unless the source establishes a revision disposition; age alone is not used to mark an otherwise valid result historical.\n"
    },
    {
      "title": "Evaluation Pi Bench Quality",
      "text": "# pi-bench evaluation import and second quality review\n\nReviewed 2026-10-11. Original repository pinned to **8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5** (HEAD unchanged on the second check). Only evaluation-pi-bench.json, model-links-evaluation-pi-bench.json and this report were written.\n\n## Delivered scope\n\n**5 original campaigns, 15 aggregate records: 10 evaluation scores and 5 mean agent-task durations.** All rows use settings.evaluation_level=overall. There are no per-task benchmark rows. Five source-specific model-link mappings cover every row. Coverage contains 15 entries, including excluded platforms/submissions and the duplicate live dashboard.\n\n| Reported campaign | Participating GPUs reported | Judge-approved /50 | Recorded test exit-zero /50 | Test collection/usage errors | Mean agent duration, ms |\n| --- | ---: | ---: | ---: | ---: | ---: |\n| RedHatAI/Qwen3.6-27B-FP8 |2|37|33|1|1,218,490.74|\n| Qwen3_6-27B-UD-Q4_K_XL_gguf, MTP tag |1|35|29|1|419,613.24|\n| Qwen3_6-27B-UD-Q4_K_XL_gguf, untagged |1|36|31|1|608,547.98|\n| Qwen3_6-35B-A3B-UD-Q4_K_XL_gguf |1|33|28|1|323,760.12|\n| gemma-4-26B-A4B-it-UD-Q4_K_XL_gguf |1|25|26|2|490,893.68|\n\nEach denominator is 50 published final task artifacts: **25 Django and 25 Sphinx**. These are the community SWE-bench Verified Mini subset, not the full 500-task SWE-bench Verified benchmark. Count units, questions, score_max, numerator, denominator and scoring method are explicit. Test errors remain in the 50-task denominator; no favorable denominator was silently substituted.\n\n## Main scoring correction\n\nThe website/README describes newer evaluations as test-grounded. The pinned [judge prompt](https://github.com/kyuz0/pi-bench/blob/8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5/src/index.ts#L476) and [final-score assignment](https://github.com/kyuz0/pi-bench/blob/8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5/src/index.ts#L540) instead let the LLM judge override test results. Raw artifacts confirm **27 disagreements across 250 tasks**: 23 judge approvals despite nonzero test exits, and 4 judge rejections despite exit zero.\n\nThe imported headline scores therefore say **pi-bench judge approval**, with evaluation_method “LLM judge with test output; overrides allowed”. They are not relabeled official SWE-bench resolved rates. All 250 eligible artifacts contain sweContainerTest=true and integer test-exit fields, so these five campaigns are test-informed judging rather than legacy judge-only campaigns.\n\nSeparate **pi-bench FAIL_TO_PASS exit-zero** scores are derived by counting published sweTestExitCode==0. They do not assert that the official full SWE-bench harness was run. [The runner](https://github.com/kyuz0/pi-bench/blob/8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5/src/index.ts#L375) executes selected FAIL_TO_PASS tests, with no PASS_TO_PASS evaluation. Test scores are marked partial because some tests did not execute successfully; the original50 records still determine their denominators.\n\nEvery campaign has a malformed/truncated parameterized test selector for sphinx-doc__sphinx-8265, producing pytest exit 4 rather than a test assertion result. Gemma additionally has an exit 4 collector error for sphinx-doc__sphinx-8638; its cause may involve the submitted code and is not automatically attributed to infrastructure. Relevant task IDs, exit-code distributions and judge/test disagreement IDs are retained in settings.\n\n## Reconstructed summaries\n\nThe two 27B GGUF directories each contain 50 original task results but summary.json contains only sphinx-doc__sphinx-9698, reporting 1/1. Those files are incomplete final-task snapshots, not evidence of 100% campaign success.\n\nAll 250 results-*.json files were downloaded and parsed at the immutable repository commit. Aggregates were reconstructed from final task files only. No -attempt files exist in these eligible directories. Reconstructed judge counts, total durations and rounded mean durations exactly match both the pinned [docs/data.json](https://github.com/kyuz0/pi-bench/blob/8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5/docs/data.json) and [live dashboard](https://pi-local-coding-bench.dev/). The 3 complete summary files agree; the rows present in the 2 incomplete summaries also match their corresponding raw task artifacts. The live and pinned dashboard metadata are identical, with generatedAt 2026-06-23T10:24:15.558Z. The live page is a duplicate representation, not another independent source.\n\n## Hardware, model and scaffold provenance\n\nActual [single-card platform.json](https://github.com/kyuz0/pi-bench/blob/8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5/benchmark_results/r9700/platform.json) names AMD R9700 AI PRO; [dual platform.json](https://github.com/kyuz0/pi-bench/blob/8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5/benchmark_results/dual-r9700/platform.json) explicitly names 2x AMD R9700 AI PRO. The README's Radeon 9700 16GB block is only an example and was not imported.\n\nCounts 1/2 describe the reported benchmark platforms, not independently instrumented device activity. Dual-card parallelization is unspecified; no TP2 claim is invented. Single-card rows use single_gpu based on the reported single platform. The platform ram values 32GB/64GB are ambiguous between host and device memory, so both host RAM and per-chip VRAM remain null. CPU, host OS, inference backend/build, context, batch, concurrency, launch flags, power and exact hardware topology also remain unknown where absent. Docker task-container details are preserved separately from the inference host.\n\nPer-run metadata records engine llama.cpp or vllm and a reported ROCm 7.2.4 string. Inference profiles are null. Global models.json defaults are not evidence that those context/max-output limits applied to each run. The MTP-tagged 27B campaign has no published draft depth; an untagged run is not proof of speculation being disabled.\n\nMetadata prefixes from all 250 original transcripts were examined without downloading their full 64 MB contents. Every dual-card transcript identifies RedHatAI/Qwen3.6-27B-FP8 and provider vllm. All single-card transcripts identify only local-model and provider llama.cpp. Consequently, the exact sanitized GGUF run labels remain in model fields; no publisher namespace or original punctuation is guessed. Four mappings provide clearly labeled official upstream references. The RedHatAI mapping links the explicitly tested repository; its historical HF revision and weight hash remain unknown. All five model-repository destinations were verified through Hugging Face metadata.\n\nObserved task-start timestamp spans:\n\n- Dual FP8: 2026-06-14T10:56:26.109Z through 2026-06-15T14:54:47.632Z.\n- Single 27B MTP: 2026-06-05T13:21:59.637Z through 2026-06-05T20:06:29.537Z.\n- Single 27B untagged: 2026-06-05T20:31:47.224Z through 2026-06-06T05:04:49.153Z.\n- Single 35B: 2026-06-06T07:05:46.810Z through 2026-06-06T11:42:47.955Z.\n- Single Gemma: 2026-06-06T12:52:51.498Z through 2026-06-06T19:36:33.879Z.\n\nreported_at uses the earliest observed task date, explicitly labeled as such; it is not a publication date. The last timestamp is a last task START, not a claim that the whole campaign ended then. Source publication dates are unknown.\n\nThe agent is pi-coding-agent, orchestrated by pi-bench with a task-instruction preamble and coding tools. Five full sample transcripts and the runner confirm this preamble; the import does not present an entirely unwrapped agent or claim that its system prompt was modified. Exact run-time agent and harness versions are not recorded. Repository package 1.0.0 and lockfile agent 0.72.1 are stored as snapshot references only. Likewise, the repository's 50 task files and Git blob IDs provide a review snapshot, not proof of the historical HF dataset revision.\n\nThe task-definition manifest SHA256 is 39bbafce0836cac0734dcd758aaf781b41cc94a918f34b3fefb14c6b296581dc. Each row also records a deterministic manifest hash of the 50 original raw-result file hashes. Task membership matches across all five campaigns.\n\nOne final result per task is not sufficient to prove the configured retry/pass@k policy. pass_k and attempts remain null; observed_final_results_per_task=1 and absence of retained attempt histories are explicit. Internal agent retries/re-prompts are not recast as independent pass@k samples.\n\n## Timing boundary\n\n[durationMs is captured](https://github.com/kyuz0/pi-bench/blob/8fbd7c6015a1ebaf1fd1d2bf257d066106aa3bb5/src/index.ts#L367) after agent execution and possible harness re-prompts, before the test and judge phases. The five duration rows are arithmetic means over 50 published values. They include agent tool work and are neither decoder speed nor full end-to-end evaluation-job duration. No token/s estimate is derived from these times.\n\n## Distinct second quality pass\n\n- Re-read HEAD and confirmed the pinned commit is unchanged.\n- Checked 250 raw task artifacts: unique task membership, binary judge scores, all test fields present, finite positive durations, no duplicated attempt files.\n- Recomputed all 15 imported numerical values independently from retained raw task objects; no mismatches.\n- Matched the 50 task IDs to the 50 task-definition files and checked all five campaigns use the same set.\n- Compared raw artifacts to each summary; found only the two incomplete 1-task summaries, with no altered values in the overlapping entries.\n- Compared all eligible model counts/durations against the repository and live dashboard; exact agreement apart from the dashboard's documented integer mean rounding.\n- Verified all 250 transcript model/provider/timestamp prefixes and examined five full transcript samples for scaffold provenance.\n- Validated 15 unique row IDs, 5 unique sources, complete required fields, overall evaluation level, correct 50-task score denominators and exactly one source/model link mapping for every row. No validation errors.\n\n## Exclusions and remaining limits\n\nStrix Halo and dual-Strix-Halo campaigns are APUs outside scope. Cloud agent-provider examples are not local PRO results; the externally hosted Gemini judge remains metadata for the local-agent evaluation. All 6 open PRs were inspected at the listing level: PR3/5/6/9 explicitly submit Strix Halo results, PR7 changes container support only. All 589 changed paths in PR8 were checked; result platforms are NVIDIA RTX 5060 Ti and Strix Halo, with no eligible PRO additions. These result exclusions are recorded in coverage.\n\nAll eligible aggregate campaigns found in this pinned source were imported. Missing inference launch settings, exact GGUF provenance, retry budgets and official full-harness resolved rates remain genuine gaps. They cannot be repaired by inheriting README examples, current default configurations or other hardware's settings. No benchmark was rerun, and this review does not claim independent hardware replication.\n"
    },
    {
      "title": "Evaluation Source Navigation Quality",
      "text": "# Source navigation and benchmark catalog review — 2026-10-11\n\n- The reported run `run-4bc8571867d2af67ea2c` correctly contains 45 measurements across 18 model/configuration combinations and three models. Its Pass A artifact is `data/qualification/results.json` at NikoCloud commit `ec80a08d7a1e506a3310efb5f67c8eaf31800b7f`. The broader campaign has six files and 352 measurements. The apparent dead link was navigation to the already selected run.\n- Ninety of 108 source campaigns contain only one run group. The interface now suppresses redundant run/source scope actions and labels external report navigation explicitly.\n- Restored explicit raw evidence links for 178 measurements: 78 evaluation rows with absolute `raw_file` web URLs and 100 Tosh rows with `raw_attachment_url`. Relative repository files still require a full commit before constructing a permalink. Explicit attachment URLs are not claimed to be immutable merely because they are links.\n- Corrected the report destination for `eval-agent-0051` through `eval-agent-0058`. Their `settings.report_url` explicitly points to `benchmarks/quality/tooluse256k-qwen38-27b-fp8-v0518-r9700.json` at mattbucci commit `9d43d97db04df090d76bad8ff813a96c3189c198`. The old campaign link pointed to the separate canonical quality file. The existing eight-record run ID `run-8929122e038008ab10a7` remains unchanged. A run uses a member report override only when all member report URLs agree; otherwise its header retains the campaign URL.\n- No source/run foreign-key or count inconsistency was found. The source audit checked 1,203 pinned GitHub destinations against 17 repository revision trees; those paths exist. This is a repository-path check, not a claim that every external website was live-checked in a browser.\n- Benchmark family and version are browsing fields for both quality and performance measurements. Engine versions and checkout commits do not fill missing benchmark versions. MBPP base/expanded tests share the MBPP+ v0.2.0 corpus but retain separate test sets. Terminal-Bench-Local uses Core19 v1.0.0; upstream Terminal-Bench 2.1 remains provenance. Tool-use receipt schema-v2 is not a benchmark version. pi-bench snapshot package/lockfile versions are not attributed to historical runs.\n- All 9,917 measurement IDs, numeric values, units, exact models, GPU assignments and run IDs are preserved. Only reviewed evidence-link corrections and derived browsing metadata changed. There are still 155 evaluation rows, including 61 subtests, and 9,762 other measured metrics; these are measurements, not independent experimental replications.\n"
    },
    {
      "title": "Expansion Labs Quality",
      "text": "# AMD-Pro-Bench expansion labs — LLM-only quality pass\n\nChecked 2026-10-11. Final lane output: **111 LLM rows, 6 source objects, 15 coverage records**. Card distribution: W7900 40, W7800 69, W7600 2. Four rows use two participating GPU chips; the other 107 use one. Every benchmark row has card_id, an explicit participating-chip count, and LLM category.\n\n## Scope correction applied\n\nThe user narrowed the project to benchmarks related to LLM models. **559 out-of-scope draft rows were removed** from this expansion input. These covered gaming, rendering, general compute, image/video and unrelated suite monitoring. No such rows remain in expansion-labs.json. General GEMM/kernel microbenchmarks, theoretical hardware specifications, and CPU-only results are also excluded.\n\nThe previous research files were not edited. The root build handles filtering legacy inputs. The sole reused source object, lab-phoronix-vllm-2025, was compared with the live original labs.json and is byte-for-byte equivalent when serialized. Its new rows contain independently read W7900 column values; no R9700 measurement was copied into another card.\n\n## Distinct second source review\n\nThe second pass revisited original source sections, chart labels, source publication metadata, corrections and author replies after the numerical draft. It also independently recomputed two claimed comparisons.\n\n1. **Phoronix W7900 vLLM columns.** [Original October2025 campaign](https://www.phoronix.com/review/amd-radeon-ai-pro-r9700) supplies DeepSeek-R1-Distill-Qwen-7B, DeepSeek-R1-Distill-Llama-8B and deepseek-moe-16b-chat rows. Exact numerical SVG labels were read for the W7900 column. Total-token throughput and output-only throughput remain different metrics; latency seconds were converted to milliseconds. Only model-tied power rows remain. Suite-wide power/temperature/composite rows were removed at the LLM-only scope correction. The driver reports about45GB visible memory while the established board has48GB nominal memory; these are stored separately. The article's unpublished local vLLM profile prevents recovery of exact request shapes or the container tag, so these fields remain unknown.\n\n2. **Puget MLPerf Client subset.** [December18,2025 roundup](https://www.pugetsystems.com/labs/articles/2025-professional-gpu-content-creation-roundup/) explicitly excludes multi-GPU configurations. Only its two LLM-token metrics were imported: second-plus-token generation and geometric-mean TTFT for W7900, W7800 and W7600. The [original generation chart](https://wp-cdn.pugetsystems.com/2025/12/PGR25_MLPerf_2token.png) and [TTFT chart](https://wp-cdn.pugetsystems.com/2025/12/PGR25_MLPerf_ttft.png) were inspected. MLPerf Client supports multiple selectable models, and the reviewer does not disclose the selected model set or quantizations. The workload is therefore labeled as an LLM suite with selected models unreported, rather than inventing a Llama/Phi identity. The W7800 memory variant is unreported and stays null. The two roundup articles use different motherboards; content-creation MLPerf rows use the documented X870E setup, not the engineering X670E setup. All other application charts are excluded.\n\n3. **W7800 48GB initial and corrected llama.cpp report.** [LCZ topic1538](https://lcz.me/topic/1538) has explicit revisions rejecting its earlier Vulkan-superiority and quantized-KV recommendations. Historical configurations and measured rates remain, with their backend/quantization boundaries preserved; old advice is not presented as current. The owner-written hardware list names Ryzen5 7500F and B850I, whereas an AI-assisted section incorrectly says Ryzen7 7500F and B650M. The directly stated owner configuration is used and the conflict remains flagged. Published date was verified from original article metadata as2026-09-07; explicitly dated experiments retain their experiment dates. The claimed approximately85.4million output tokens over the capture is not consistent with the listed rates and was not imported as throughput. Its request-rate distribution is retained with a mixed-configuration/token-accounting flag.\n\n4. **W7800 48GB SGLang follow-up.** [LCZ topic1718](https://lcz.me/topic/1718), published2026-09-15 in original metadata, explicitly establishes a single70-CU Gigabyte W7800, TP=1, custom SGLang based on gfx1100-support@1442c18, and Qwen3.8-27B-INT4-GPTQ-v3. The appendix confirms running-req=1 for the principal rate, whereas max-running-requests4 is only a capacity limit. The physical device identity was checked against its appendix: gfx1100, device0x7449, Gigabyte subvendor0x1458, subsystem0x2428, Gen4x16. The11 listed generation rates have arithmetic mean **60.1764** and median **62.11**. The headline62.11 matches the median, while the earlier50.9 baseline was labeled a mean. Consequently, the claimed22% uplift was excluded and this comparison is flagged. The report also documents unresolved perplexity/GSM8K regression and greedy nondeterminism; the measured quality scores remain tied to that model and configuration. Wrong-model experiments, a HYBRID result later identified as DFLASH fallback, and generic kernel microbenchmarks were excluded. Later custom-engine success does not invalidate the older official-FP8 failure measurement; they are different software paths.\n\n5. **W7900 Dual Slot, actual multi-GPU experiment.** [David Huang's July11,2024 report](https://blog.hjc.im/dual-w7900ds-llm-preliminary-experience.html) explicitly uses two W7900 Dual Slot boards/chips. The two-slot name is not a two-GPU board. The Llama3 70B Q8_0 result has separate layer-split and row-split measurements. Original [chart labels](https://blog.hjc.im/wp-content/uploads/2024/07/llama.cpp-perf.png) were inspected: prefill263.18/108.18 and decode9.25/12.85 tokens/s for layer/row respectively. The source's command lines repeat split flags; final flag and chart legend establish the intended mode. Comparisons with other vendors were uncontrolled cloud environments, and their values were not imported.\n\n6. **W7900 single GPU, concurrency and speculative decoding.** [David Huang's November13,2024 report](https://blog.hjc.im/apple-uma-for-llms-problems.html) pins llama.cpp commit a6744e43e80f4be6398fc7733a01642c846dce1d and describes an AMD WMMA flash-attention patch. IQ4_XS microtests are separate from Q4_0 speculative decoding. Concurrency4–12 rates are aggregate and were not divided to manufacture per-user measurements. The speculative log explicitly reports16 input and **517 actual outputs**; requested512 is stored separately. The printed25.924 versus15.29 tokens/s implies **69.55%**, not the text's80% improvement. Absolute figures are retained, the relative claim is flagged and excluded as a benchmark row. Author follow-ups describe newer vLLM improvements without retracting this historical campaign. TG1024 and concurrency1/2 raster labels overlap another series and were omitted rather than guessed.\n\n## Metadata and structural checks\n\n- GPU counts represent participating chips. Two request slots, draft length8, HiCache8 and capacity for four requests do not become GPU counts.\n- Per-GPU VRAM is never a sum across cards. W7900 Dual Slot is48GB per chip; W7800 AI TOP48G is48GB per chip.\n- Quantization and KV-cache dtype are separate. Unknown quantization in A/B tables remains null.\n- Configured output/request limits stay in settings. Explicit actual log counts and fixed-length llama-bench microtests may populate actual token fields.\n- Historical marks identify documented older configurations, not automatically invalid data. No unknown revision was aged into historical merely because the page is old.\n- Relative claims, requested lengths, single-request rates, aggregate rates, suite rates and quality scores remain distinct.\n- The final structural pass found zero non-LLM categories, duplicate IDs, missing source references, missing card IDs, nonnumeric values, or single-GPU/count contradictions.\n- Primary-source date metadata was used; no invented publication days.\n\n## Coverage boundaries\n\nThis lane found no original measured W7400 LLM result. Estimated model-fit sites and unverified secondary aggregators were excluded. W7500/W7700 original repository reports were handed to the raw-data lane to avoid duplicate ingestion. W7800 32GB is not inferred from a bare W7800 label where the source does not establish memory capacity. Neither source verification nor arithmetic review is a claim of independently reproducing the benchmarks on hardware.\n\n"
    },
    {
      "title": "Expansion Owner Quality",
      "text": "# Exact model identity and rare-card evidence review\n\nReviewed 2026-10-11 after the LLM-only cleanup.\n\nThe two W6900X reports come from the same author, in different threads with different settings. The first is a comment under a post about RTX3090s and IQ3XS. The W6900X commenter does not restate the quantization or context. Those root-post values were deliberately not inherited. The later first-person post explicitly reports Qwen3.8-Flash-Next UD-Q6_K_XL, two W6900X32GB GPUs, 384GB host RAM, 32768 configured context, layer split38,12, and35 CPU MoE layers. Its12–15 decode and100–140 prefill ranges are separate endpoints, not invented averages.\n\nOriginals: https://www.reddit.com/r/unsloth/comments/1w253wh/single_gpu_faster_than_dual_gpu_for_large_moe_cpu/ and https://www.reddit.com/r/LocalLLM/comments/1w3r8y5/qwen38flashnext_performance_tuning_for_hybrid/ . Both were read again for settings. Operating-system version, concurrency, actual prompt length, checkpoint repository, and exact runtime build remain unresolved. No later author's suggestions were applied to already-reported measurements.\n\nSearches for W6600, W6400 and W7600 LLM rates surfaced estimators and compatibility lists more often than timed runs. welche.ai explicitly states W6600 speed was not measured on the card. FitMyLLM W6900X explicitly derives speed from bandwidth and model size. APXML supplies approximate throughput without a verified original run. These were added to the coverage register, not imported as hundreds of measured results. The Ollama W6600 issue is a compatibility report with no speed figure.\n\nThe upstream llama.cpp21043 host includes a W6600, but the environment explicitly hides it while measuring R9700. This is additional evidence that enumerated hardware is not participating hardware. Published card filters must follow per-run participation rather than host inventory.\n\nThe UI now puts exact model/checkpoint and exact quantization before card selection. Regression checks keep IQ3_XXS and IQ3_S distinct, preserve file-name underscores, warn on different card/memory variants, and distinguish two GPU chips from two physical Duo boards. Mixed W6800+RX6800 and Vega+W6800X Duo systems keep their mixed identity; their combined speed is not attributed to homogeneous workstation cards.\n"
    },
    {
      "title": "Expansion Rare Quality",
      "text": "# Rare-card LLM coverage sweep — 2026-10-11\n\nThis bounded final pass searched for original, measured LLM inference/serving/training results on W6600, W6400, W6300, W6600M, W6500M, W6300M, R9600 and R9700S. Twenty focused web queries included GitHub, Reddit and Hugging Face, exact board names, llama-bench, llama.cpp, Ollama and token-rate terms. It produced **0 eligible benchmark rows**, **4 source-only records**, and **8 coverage records**. This is a documented coverage gap, not proof that no such results exist anywhere.\n\nNo existing campaign or catalog file was changed. The website should retain these cards with zero measured LLM results instead of filling their rows from estimates or another card.\n\n## Card gaps\n\n| Card | Result of focused sweep |\n| --- | --- |\n| W6600 | Original Ollama issue and logs found, including exact Meta-Llama-3-8B-Instruct / Q4_0 metadata, but no eligible measured speed. |\n| W6400 | Architecture/compatibility aliases and prediction pages, no measured model run found. |\n| W6300 | Specification/compatibility pages and predictions, no measured model run found. W6300 also names an unrelated Ethernet controller, creating irrelevant search matches. |\n| W6600M | Workstation inventory, hardware specification and architecture-list hits, no measured model run found. |\n| W6500M | Architecture-list/compatibility hits, no measured model run found. |\n| W6300M | Architecture-list/compatibility hits, no measured model run found. |\n| R9600 | Official product and support pages plus launch discussion, no measured model run found. |\n| R9700S | Token-rate calculator and many plural-R9700 matches, no measured result for the distinct S product found. |\n\n## Second quality pass and rejected numeric leads\n\n1. Read the [Ollama issue 3825](https://github.com/ollama/ollama/issues/3825), then independently fetched and examined the complete comments through GitHub's public API. The W6600 author's logs expose Meta-Llama-3-8B-Instruct, Q4_0 and initialization failures. A later mixed W6600/W6800 report gives no usable model timing. The [approximately 6 tokens/s comment](https://github.com/ollama/ollama/issues/3825#issuecomment-2092105762) is by another participant, whose surrounding comments identify a Radeon RX 6600 XT; that participant suspects CPU fallback because allowing/removing GPU device access gives the same speed. It is **not a W6600 benchmark**.\n\n2. Re-read the [Ollama architecture wiki](https://github.com/likelovewant/ollama-for-amd/wiki/AMD-GPU-Arches-lists-Info/66716f070f398e6609f7ab393b4903cc46f5a6eb). It groups professional cards with RX cards under gfx1032/gfx1034. These are architecture aliases, not evidence that any consumer-card run was performed on a listed workstation model.\n\n3. Opened and searched the [Hugging Face GPU-spec CSV](https://huggingface.co/spaces/awacke1/Deepseek-HPC-GPU-KEDA/blob/main/data/gpu_specs.csv). The file contains GPU specifications and dates, not measured LLM execution. Its enclosing Deepseek project title is not evidence of a benchmark for each listed GPU. Its W6600X nominal date also conflicts with the primary Apple evidence already documented in the catalog campaign; no catalog dates were changed.\n\n4. Opened and rechecked the [R9700S GPU TPS page](https://gputps.com/gpus/3135-amd-radeon-ai-pro-r9700s-32gb). Its opening explanation and footer expressly describe the numbers as estimates from bandwidth and model size. No values were imported. Other search results use the plural “R9700s”; those belong to R9700, not the separate R9700S SKU.\n\n5. Read the [R9600D announcement discussion](https://www.reddit.com/r/LocalLLaMA/comments/1taa74c/powercolor_launches_radeon_ai_pro_r9600d_with/). Its 127 to 138.5 tokens/s comment explicitly concerns an R9700 and omits model/checkpoint and quantization. It is not evidence for R9600 or R9600D and was not imported.\n\n6. The previous original-source review of [llama.cpp discussion 21043](https://github.com/ggml-org/llama.cpp/discussions/21043) already established that W6600 was hidden while R9700 GPUs ran the tests. This sweep encountered it again but did not duplicate those results or reinterpret them as W6600 measurements.\n\n## Additional primary catalog evidence, for owner review\n\nThe [AMD GPUOpen developer-tool announcement](https://gpuopen.com/learn/radeon-developer-tool-suite-shader-source-code/) is explicitly dated **June 11, 2026** and names the **Radeon AI PRO R9600 Series**. This provides family-level existence evidence earlier than the catalog's August 17, 2026 driver citation naming individual products. It does not distinguish the R9600/R9600D first-shipment dates. The [R9600 product page](https://www.amd.com/en/products/graphics/workstations/radeon-ai-pro/ai-9000-series/amd-radeon-ai-pro-r9600.html) labels the image an artist render and says it is not available for purchase. Retaining the catalog's date-qualified treatment is appropriate; do not promote either reference into a confirmed retail launch date.\n\n## Verification\n\nThe JSON was parsed after writing. It contains no benchmark rows, no fabricated model or quantization fields, and no estimated rates promoted into measured data. All source IDs are unique. Coverage names all eight requested target cards; R9600D appears only as a related false-positive check.\n\n"
    },
    {
      "title": "Expansion Raw Quality",
      "text": "# AMD-Pro-Bench expansion: original raw-data quality review\n\nReviewed 2026-10-11. This bounded lane writes only expansion-raw.json and this report. It adds **4084 LLM metric records from 30 source campaigns**, with 46 coverage entries. A metric record is one measured numerical cell, not an independent benchmark run or independently replicated experiment.\n\n## Scope and card coverage\n\n- w7800: 446 records\n- w7900: 116 records\n- w7700: 6 records\n- r9700: 3387 records\n- w6800x-duo: 107 records\n- w7500: 2 records\n- w6800: 20 records\n\nOnly model-related LLM inference, serving, agent execution, model quality and model-tied memory/power are retained. Initially extracted CFD/kernel-only records were removed after the user's scope correction. Gaming, rendering, video generation, speech-only workloads, theoretical GPU specifications and memory-fit estimates are excluded. All card IDs were checked against the 22-product catalog. Absence in this lane means a coverage gap, not proof that no benchmark exists or a card cannot run LLMs.\n\n## Distinct second review and numerical checks\n\nAfter initial extraction, revisited configuration documents, raw result files, author follow-ups, current repository heads and corrections. Unknown values remain null. Historical measurements are not marked superseded merely because a newer build exists.\n\n- All 4,084 records: required-field presence, finite numerical values, unique IDs and source/cell keys, source foreign keys, LLM-only scope, valid card IDs, positive integer participating GPU counts, single_gpu/count consistency, integer token/count fields and plausible dates. **Zero structural errors.**\n- W7800: independently matched all 429 CSV-derived numerical values to original CSV cells. Zero mismatches. Later corrected the Muse addendum's GPU count and engine version before final save.\n- VTE: independently matched all 63 load/decode/TTFT values to the correct model block in eight original JSON files. Zero mismatches.\n- NikoCloud: independently traversed and checked 352 qualification numerical fields. Zero mismatches.\n- ToshLLM: 100 attachment table values independently rechecked; no numerical mismatches.\n- llama.cpp discussion 21043: independently parsed HTML table cells within the four original comment boundaries and compared all 183 imported values. Zero mismatches. Repeated summary/baseline columns were not imported twice.\n- Kai Bennett: all three original log SHA-256 hashes match the published evidence manifest. Independently parsed 552 request timing triplets (108 + 409 + 35), reproduced generated totals of 336,107 / 196,543 / 119,627 and the published rounded medians. All 2,541 imported request numerical cells were rechecked from their exact original log lines with zero mismatches. Six nonzero-depth sweep values retain separate capture identity.\n- Quant Lab: all 12 full-model normalized medians were recomputed from the five samples after two warmups; raw-file SHA-256 matches each normalized record. All 180 V4 run numerical cells were independently checked against ANALYSIS.json. All 18 pairs remain, including the adverse pair-17 excursion.\n- Re-read current GitHub commit heads for BoJl4apa, VTE, NikoCloud, ToshLLM, Kai Bennett and Quant Lab: the snapshots used here still match current heads at review time. Original issue comments/updates were revisited; geerlingguy issue35's four comment timestamps did not change during review.\n\nThe above checks validate transcription and metadata interpretation. They do not independently rerun GPU benchmarks, prove honest reporting, or make differing workloads comparable.\n\n## Resolved metadata problems\n\n### W7800 campaign versus Muse addendum\n\n[August campaign](https://github.com/BoJl4apa/rdna3-llm-bench/blob/21f4c156420982f20d3c67cb2f19c853d10a9d21/results/2026-08-unified-final.md) uses two 48 GB W7800s, layer split, Ollama 0.32.5-rocm. [Muse addendum](https://github.com/BoJl4apa/rdna3-llm-bench/blob/21f4c156420982f20d3c67cb2f19c853d10a9d21/results/2026-08-muse-glimmer.md) explicitly uses ONE card and Ollama 0.32.9-rocm for both Muse and the Qwen rerun. Corrected 78 CSV-derived rows and created a separate source; the addendum has 80 records including two quality scores.\n\nThe source's exact hardware wording claims 32 CUs with 48 GB. [AMD's official ROCm table](https://rocm.docs.amd.com/en/docs-6.4.0/reference/gpu-arch-specs.html) lists 70 CUs for W7800 48 GB. Preserved the reported GPU name and 48 GB variant, flagged the conflict, and assigned medium source strength. It is not safe to silently repair the source's physical hardware claim.\n\nThe campaign says its TTFT includes reasoning before visible answer content. Accordingly 143 CSV latency rows use time_to_first_content_ms, retaining the original field name in settings. The September MTP report labels TTFT without restating its boundary; its two rows use reported_ttft_ms and carry uncertainty flags instead of being pooled with strict TTFT. September MTP versus plain aliases and manifest digests are distinct. One quality-score difference comes from a capped, empty answer, which is preserved in notes.\n\nCampaign Q4_K_M precision is inherited only where stated and flagged as inherited; unresolved Ollama aliases remain unresolved. The Mistral-small-4-119b UD-Q4_K_M exception is preserved. October 0.35 commentary does not invalidate the August or September historical measurements. Large-model failures attributed to host RAM occupied by VMs are not recast as engine regressions.\n\n### GPU selection, chips, boards and mixed systems\n\n- [NikoCloud](https://github.com/NikoCloud/r9700-local-inference-notes/blob/ec80a08d7a1e506a3310efb5f67c8eaf31800b7f/docs/07-model-qualification.md): explicit Vulkan0 selects one R9700. An installed RX9070XT is not counted. The 1,915.5 tokens/s at 60 streams headline belongs to RX9070XT with a 9B model and is excluded. Server -np capacity is distinct from actual concurrent requests.\n- [ToshLLM issue16](https://github.com/engeldlgado/toshllm/issues/16): four participating W6800X Duo chips are not four boards. Per-chip VRAM is 32 GB. Later three-chip narrative has contradictory physical-board wording; retain count three from explicit selected devices, flag the narrative conflict and leave board count unknown. Its split algorithm is unknown, so corrected layer_split to multi_gpu_unspecified.\n- [ToshLLM issue49](https://github.com/engeldlgado/toshllm/issues/49): two selected Duo chips, explicit layer split. W6800X Duo has two GPU chips per board, as confirmed by [Apple](https://support.apple.com/en-la/101868), but physical board count is not inferred for these selected devices. Peer transport was ON on both compared versions; the maintainer corrected the earlier A/B interpretation. The common upstream build hash does not imply the Tosh binaries are identical.\n- [W6800 Nemotron original](https://www.reddit.com/r/LocalLLaMA/comments/1prxpcx/nvidia_nemotron3nano30b_llm_benchmarks_vulkan_and/): Q8_0 measurements use one W6800 plus one RX6800. gpu_count=2 is total participating chips, with both cards recorded. Top-level per-chip VRAM is null for this mixed system. They must not be rendered as two W6800s. Q5_K_S tests use W6800 alone. Do not inherit OS from the author's older Llama2 post.\n- Modded V620 described as W6800 in Reddit post1u7x7zh is explicitly excluded from the retail W6800 corpus. The two imported EmPips W6800 posts do not make that claim.\n- [llama.cpp 21043](https://github.com/ggml-org/llama.cpp/discussions/21043): W6600 is installed but explicitly hidden. March/April tables use one or two R9700s; -sm row and later -sm layer are distinct. This produces no W6600 benchmark record.\n- [Quant Lab](https://github.com/1337hero/r9700-quant-lab/blob/e35bd52cbf37681a9bc630b3519cf6a602781f2f/config/hardware.yaml): three R9700s installed. Full-model llama-bench explicitly selects ROCm2 with -sm none, therefore ONE GPU despite three names in gpu_info. V4 agent and resident-context tests use TWO non-display cards with explicit layer split. MBPP assigns each model ONE separate card concurrently; neither two-card model parallelism nor cross-card timing comparability is implied.\n\n### Source follow-ups and comparability\n\n- [VTE W7900](https://github.com/kyuubyN/VTE/blob/07e1396f6e071698171e285cc03d1ca29cd62220/docs/BENCHMARK_W7900.md): VTE applies a chat template while llama-cpp-python uses raw completion. Preserve prompt_format_mismatch. Some raw completion counts are 151 despite a configured cap of150; retain the actual count and flag the conflict. The VTE maintainer benchmarks their own engine, so source independence is unclear despite high raw reproducibility.\n- [RWKV issue14](https://github.com/RWKV-Vibe/RWKV-Inference-Performance-Test/issues/14): heading says ROCm but actual adapter log is Vulkan. Use Vulkan and retain the discrepancy. Intel UHD730 rows are excluded. RPC,Vulkan in issue9 denotes loaded backends, not evidence of remote or multiple GPU execution.\n- [geerlingguy issue35](https://github.com/geerlingguy/ai-benchmarks/issues/35): keep PiCM5, Intel265K and later FA/mmap runs separate. The author questioned initial Pi GPT-OSS results. Qwen3-30B-A3B heading conflicts with dense32B raw metadata and is flagged. Power boundary here is unreported; issue1's explicitly whole-system outlet power is not inherited into issue35.\n- [NixOS PR488117](https://github.com/NixOS/nixpkgs/pull/488117): W7500-only Qwen2.5-3B-Instruct results are actual report values. Example Q4_K_M,8192 context and proposed build numbers are not proof of timed settings; actual unknown quantization/context/build remain null. MI50 is installed separately and does not participate. Discourse copy is a duplicate.\n- [Wendi-2B](https://huggingface.co/wenzani/Wendi-2B-GGUF/blob/85f35499daa41efe773988c10c160eb0f28df8fa/README.md): next-token option-probability decisions, not autoregressive decode speed. Unknown GPU count remains null. GIF playback rate is excluded. Cache controls and decision-latency scopes are explicit. The model author is marked vendor.\n- [Original W7900 multi-engine project](https://llm-tracker.info/W7900-Pervasive-Computing-Project): card supplied through AMD contest is disclosed. vLLM throughput counts input PLUS output across1,000 prompts and remains aggregate; ExLlama and llama.cpp generation are separate metrics. MLC memory compiler estimates, unnamed bitsandbytes70B results, NVIDIA comparison, kernel-only and speech measurements are excluded. Historical build versions are retained.\n\n## Additional final-pass corrections from new raw repositories\n\n[Kai Bennett's evidence guide](https://github.com/KaiFelixBennett/local-ai-amd-benchmark/blob/d81f5f2a46c06118b3a8aec5aa2e152c7688320d/evidence/README.md) defines a 200-output-token floor for decode statistics; below it timing resolution can produce extreme apparent speeds. The import omits 219 short-response decode rates, retaining their observed prefill and latency. Per-request internal model timing is not end-to-end agent wall time. All copied Halo/cloud results are excluded. The longer Q4 agent run is only partially logged, so log totals are not whole-job totals. Qwen3.6's author-disqualified game-quality score is not imported; the source expressly retains its speed evidence.\n\nThe original 35-task report's26.43 decode median differs from30-qualified-response26.30. Both are mathematically explained by the sample filter; they are not contradictory card measurements. This import takes the raw request data and does not add those derived summary rows again. The log-backed depth sweeps retain UD-Q6_K_M and UD-Q4_K_XL exactly. Companion CSVs have no numeric rows; data comes from their .log files. Cold first points and unmeasured MTP projections are excluded. Default split_mode=layer in a one-device bench does not become multi-GPU parallelism.\n\n[Quant Lab's final V4](https://github.com/1337hero/r9700-quant-lab/blob/e35bd52cbf37681a9bc630b3519cf6a602781f2f/reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/end-to-end/scaffold-v4/20260714T172715Z/SUMMARY.md) supersedes the tempting rejected V3 headline as the final interpretation. The 5.82% equal-work gain and2.21% raw wall-time gain are different measurements; the latter fails the fixed3% effect threshold. V4's adverse pair17 remains in data. No claim that agent jobs reliably finish faster is made. Model quality was generated separately from CPU evaluation; only the generation card count is represented. Context262144 is capacity, not an actual prefed prompt. Memory numbers are resident snapshots, not peaks. The custom Q8_0_ROCMFPX integer format is neither native FP8 nor ordinary Q8_0. Developer affiliation is marked unclear independence.\n\nFor consistent filtering, 352 lowercase GGUF quant labels were normalized to uppercase while preserving the original string in settings.reported_quantization. No precision formats were merged. Exact model filenames and known artifact hashes remain distinct, and unresolved aliases retain warning flags.\n\n## Source inventory\n\n- [exp-bojl-w7800-aug](https://github.com/BoJl4apa/rdna3-llm-bench/blob/21f4c156420982f20d3c67cb2f19c853d10a9d21/results/2026-08-per-leg.csv): 360 cells; medium strength; raw_results.\n- [exp-vte-w7900](https://github.com/kyuubyN/VTE/blob/07e1396f6e071698171e285cc03d1ca29cd62220/docs/BENCHMARK_W7900.md): 63 cells; high strength; raw_results.\n- [exp-bojl-w7800-mtp](https://github.com/BoJl4apa/rdna3-llm-bench/blob/21f4c156420982f20d3c67cb2f19c853d10a9d21/results/2026-09-qwen3.8.md): 6 cells; medium strength; raw_results.\n- [exp-rwkv-w7900-llama](https://github.com/RWKV-Vibe/RWKV-Inference-Performance-Test/issues/9): 8 cells; high strength; raw_results.\n- [exp-rwkv-w7900-web](https://github.com/RWKV-Vibe/RWKV-Inference-Performance-Test/issues/14): 6 cells; high strength; raw_results.\n- [exp-geerling-w7700-pi](https://github.com/geerlingguy/ai-benchmarks/issues/1): 6 cells; high strength; raw_results.\n- [exp-geerling-r9700-platforms](https://github.com/geerlingguy/ai-benchmarks/issues/35): 65 cells; high strength; raw_results.\n- [exp-niko-qualification](https://github.com/NikoCloud/r9700-local-inference-notes/blob/ec80a08d7a1e506a3310efb5f67c8eaf31800b7f/docs/07-model-qualification.md): 352 cells; high strength; raw_results.\n- [exp-niko-pld-concurrency](https://github.com/NikoCloud/r9700-local-inference-notes/blob/ec80a08d7a1e506a3310efb5f67c8eaf31800b7f/data/core19/pld_concurrency.json): 36 cells; high strength; raw_results.\n- [exp-tosh-w6800x-duo](https://github.com/engeldlgado/toshllm/issues/16): 2 cells; high strength; raw_results.\n- [exp-tosh-w6800x-three-chips](https://github.com/engeldlgado/toshllm/issues/16#issuecomment-4885667733): 5 cells; medium strength; partial_method.\n- [exp-hf-wendi-w7900](https://huggingface.co/wenzani/Wendi-2B-GGUF/blob/85f35499daa41efe773988c10c160eb0f28df8fa/README.md): 4 cells; medium strength; detailed_method.\n- [exp-tosh-duo-attachments](https://github.com/engeldlgado/toshllm/issues/49): 100 cells; high strength; raw_results.\n- [exp-nixos-w7500](https://github.com/NixOS/nixpkgs/pull/488117): 2 cells; medium strength; partial_method.\n- [exp-reddit-w6800-llama2](https://www.reddit.com/r/LocalLLaMA/comments/1pob44f/32gb_mi50s_were_getting_so_expensive_that_i_ended/): 8 cells; medium strength; partial_method.\n- [exp-reddit-w6800-nemotron](https://www.reddit.com/r/LocalLLaMA/comments/1prxpcx/nvidia_nemotron3nano30b_llm_benchmarks_vulkan_and/): 12 cells; medium strength; partial_method.\n- [exp-zed-r9700-single](https://github.com/ggml-org/llama.cpp/discussions/21043#discussioncomment-16348282): 42 cells; high strength; detailed_method.\n- [exp-zed-r9700-dual](https://github.com/ggml-org/llama.cpp/discussions/21043#discussioncomment-16358718): 93 cells; high strength; detailed_method.\n- [exp-zed-r9700-april](https://github.com/ggml-org/llama.cpp/discussions/21043#discussioncomment-16521552): 32 cells; high strength; detailed_method.\n- [exp-zed-r9700-layer](https://github.com/ggml-org/llama.cpp/discussions/21043#discussioncomment-16521727): 16 cells; high strength; detailed_method.\n- [exp-bojl-w7800-muse](https://github.com/BoJl4apa/rdna3-llm-bench/blob/21f4c156420982f20d3c67cb2f19c853d10a9d21/results/2026-08-muse-glimmer-per-leg.csv): 80 cells; medium strength; raw_results.\n- [exp-lhl-w7900-original](https://llm-tracker.info/W7900-Pervasive-Computing-Project): 35 cells; high strength; raw_results.\n- [exp-kai-qwen38-27b-q4xl-moorhuhn-r9700](https://github.com/KaiFelixBennett/local-ai-amd-benchmark/blob/d81f5f2a46c06118b3a8aec5aa2e152c7688320d/evidence/logs/qwen38-27b-q4xl-moorhuhn-r9700.log): 514 cells; high strength; raw_results.\n- [exp-kai-qwen36-27b-q6-moorhuhn-r9700](https://github.com/KaiFelixBennett/local-ai-amd-benchmark/blob/d81f5f2a46c06118b3a8aec5aa2e152c7688320d/evidence/logs/qwen36-27b-q6-moorhuhn-r9700.log): 1857 cells; high strength; raw_results.\n- [exp-kai-qwen38-27b-q6-clairobscure-r9700](https://github.com/KaiFelixBennett/local-ai-amd-benchmark/blob/d81f5f2a46c06118b3a8aec5aa2e152c7688320d/evidence/logs/qwen38-27b-q6-clairobscur-r9700.log): 170 cells; high strength; raw_results.\n- [exp-kai-r9700-depth](https://github.com/KaiFelixBennett/local-ai-amd-benchmark/blob/d81f5f2a46c06118b3a8aec5aa2e152c7688320d/evidence/reports/qwen38-27b-decode-depth-q6-vs-q4.md): 6 cells; high strength; raw_results.\n- [exp-quantlab-full-model](https://github.com/1337hero/r9700-quant-lab/blob/e35bd52cbf37681a9bc630b3519cf6a602781f2f/reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/full-model/block-a/01-custom/normalized.json): 12 cells; high strength; raw_results.\n- [exp-quantlab-v4](https://github.com/1337hero/r9700-quant-lab/blob/e35bd52cbf37681a9bc630b3519cf6a602781f2f/reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/end-to-end/scaffold-v4/20260714T172715Z/ANALYSIS.json): 182 cells; high strength; raw_results.\n- [exp-quantlab-context-memory](https://github.com/1337hero/r9700-quant-lab/blob/e35bd52cbf37681a9bc630b3519cf6a602781f2f/reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/deployment/262144-two-card/SUMMARY.md): 6 cells; high strength; raw_results.\n- [exp-quantlab-mbpp](https://github.com/1337hero/r9700-quant-lab/blob/e35bd52cbf37681a9bc630b3519cf6a602781f2f/reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/quality/mbppplus/SUMMARY.md): 4 cells; high strength; raw_results.\n\n## Remaining extraction queue\n\n- [NikoCloud additional R9700 vLLM and Core19 data](https://github.com/NikoCloud/r9700-local-inference-notes/tree/ec80a08d7a1e506a3310efb5f67c8eaf31800b7f/data/vllm-mxfp4): A/B/C qualification andPLD concurrency are imported. Additional model-linked vLLM/Core19 tables remain pending; RX9070XT-only9B fanout is outside card scope.\n- [Mixed AMD LLM benchmark corpus](https://github.com/kr4ckhe4d/local-llm-benchmarks): RX9070/R9700 combinations need full per-model GPU allocation proof before adding them.\n- [Additional dp-craft serving and quality campaigns](https://github.com/dp-craft/r9700): Earlier raw lane imported exact llama-bench arrays. Remaining alias-based servingJSONL andquality campaigns need resolved checkpoint/runtime join and deduplication.\n- [ROCm original scoreboard and W7900 comments](https://github.com/ggml-org/llama.cpp/discussions/15021): W7900 llama2Q4_0 andlater original comments identified; numerical cells not imported inthis expansion yet.\n- [Kai R9700 remaining summary parameter/quant/memory tables](https://github.com/KaiFelixBennett/local-ai-amd-benchmark/blob/d81f5f2a46c06118b3a8aec5aa2e152c7688320d/evidence/reports/qwen38-27b-rdna4-quant-eval.md): Core raw agent/depth logs collected; additional August24 short-prompt/VRAM and older Qwen3.6 measurements need individual source mapping. Estimated MTP extrapolations and invalid cold depth-zero speeds excluded.\n- [llama.cpp 21043 later MTP, power-state and direct-P2P reports](https://github.com/ggml-org/llama.cpp/discussions/21043): 183 zedbytes March/April cells collected. Later Qwen3.6 MTP sweeps, dock n_ubatch dead-band measurements, direct-P2P matched Qwen2.5/Huihui follow-ups and Windows proprietary observations remain. Avoid copying repeated earlier baselines.\n- [Quant-lab earlier valid controls and archived model campaigns](https://github.com/1337hero/r9700-quant-lab/blob/e35bd52cbf37681a9bc630b3519cf6a602781f2f/reports/INDEX.md): Final model-level normalized Q8, V4 agent, MBPP and memory gates imported. Earlier valid controls require separate review; archived location alone does not imply invalidation. Kernel-only data, rejected V3 speedup headline and allocation-smoke speeds excluded.\n\nThe legacy Codeberg/StillDeadcode corpus was already imported by the earlier raw lane and is not copied again. This review does not claim the entire internet has been exhausted. Cross-lane originals were compared to avoid importing the same measurement via a mirror or recap; independent runs in the same issue or on the same model remain separate when configuration, date or raw capture identity differs.\n"
    },
    {
      "title": "Expansion Vendor Quality",
      "text": "# AMD-Pro-Bench: catalog and LLM-only expansion quality pass\n\nChecked October 11,2026. Final files: `cards.json` contains 22 base products and 11 exclusion groups; `expansion-vendor.json` contains 29 LLM records from 15 sources and 11 coverage entries. The29 records comprise 14 ToshLLM runs with separate prefill/decode metrics plus one AMD performance-per-dollar claim. All 200 earlier non-LLM expansion candidates were removed following the explicit scope correction. Legacy campaign files were not changed.\n\n## Hardware catalog verification\n\nThe [AMD professional comparison table](https://www.amd.com/en/products/specifications/professional-graphics.html) is embedded as JSON in its page HTML. It was extracted and then independently fetched again. Product names, memory and architecture were checked against that primary data. The table is incomplete: it omits desktop W6300 and Apple MPX products and leaves many current launch-date fields empty. An empty field remains unknown.\n\n- W6800, W6600 and W6600M: AMD gives June 8,2021 as launch. The [announcement](https://ir.amd.com/news-events/press-releases/detail/1008/new-amd-radeon-pro-w6000-series-workstation-graphics-with-amd-rdna-2-architecture-and-massive-32gb-of-memory-to-power-demanding-architectural-design-and-media-workloads) separately says W6800 available then, W6600 expected Q3, and mobile OEM systems expected July. Those dates are not conflated.\n- W6400, W6500M and W6300M: January 19,2022 is the [announcement](https://ir.amd.com/news-events/press-releases/detail/1043/new-amd-radeon-pro-w6000-series-graphics-unleash-high-efficiency-powerful-cad-performance-for-mainstream-workstation-users) and AMD table launch date. Physical availability was described as Q1 or later2022, not necessarily that day.\n- W7900/W7800: April 13,2023 announcement; expected retail Q2. W7500/W7600: August 3,2023 announcement explicitly says available that day. W7700: November 13,2023 announcement and expected availability; a first-ship datasheet is now404, so first shipment is not asserted.\n- W7900 Dual Slot remains a48 GB variant of W7900, with [June 19,2024 expected availability](https://www.amd.com/en/newsroom/press-releases/2024-6-2-amd-unveils-next-gen-zen-5-ryzen-processors-to-p.html). Dual Slot means board width, not two GPUs. W7800 has32 GB and 48 GB variants; both use one GPU. AMD's [ROCm architecture table](https://rocm.docs.amd.com/en/docs-6.4.0/reference/gpu-arch-specs.html) states70 CUs for the 48 GB model. A community32CU label is therefore a source discrepancy, not a reason to rewrite the catalog. W7900D is a named official-driver variant with unverified detailed configuration.\n- Apple W6800X/W6900X/W6800X Duo have an [August 3,2021 availability announcement](https://www.amd.com/en/newsroom/press-releases/2021-8-3-new-amd-radeon-pro-w6000x-series-gpus-bring-ground.html). [Apple confirms](https://support.apple.com/en-la/101868) one Duo module contains two GPU chips with 32 GB each. Catalog memory is per chip:32 GB, not64 GB. Two Duo boards may provide four chips, but benchmark participation still needs evidence. W6600X8 GB is present in the [March 11,2022 Apple price list](https://www.apple.com/education/pricelists/pdfs/Apple_US_Education_Institution_Price_List-03-11-2022.pdf). Its March 8 value is a **pricing date**, not automatically a first-shipment date; launch_date remains null.\n- R9700 announcement corrected during the second pass to [May 20,2025](https://newsroom.amd.com/news/amd-introduces-new-radeon-graphics-cards-and-ryzen/); OEM workstation availability is separately July 23. R9700S, R9600D and R9600 each have one GPU chip and 32 GB; their active/passive and single/dual-slot distinctions are preserved.\n\nSeven entries have `scope_status: date_qualified`: W6300, W7400, W7500M, W6600X, R9600, R9700S and R9600D. Their `scope_reason` explains what is verified and why current product listings are not being presented as shipping proof. `first_verified_by` is anchored to dated primary evidence, not an inferred launch:\n\n- W6300: [AMD PRO 22.Q4](https://www.amd.com/en/resources/support-articles/release-notes/RN-PRO-WIN-22-Q4.html), updated November 14,2022; OEM board specifications additionally come from Dell. Release year stays unknown.\n- W7400: [AMD Adrenalin 26.9.2](https://www.amd.com/en/resources/support-articles/release-notes/rn-rad-win-26-9-2.html), September 29,2026. AMD's overview says MiniDP1.4 while its detailed field says MiniDP2.1; the mismatch is documented.\n- W7500M: [Panasonic March 12,2026 announcement](https://eu.connect.panasonic.com/be/en/news/new-toughbook-56-reimagines-rugged-mobility-usability-field-workers) names the 8 GB option and expected May systems. This does not identify the GPU's first shipment.\n- R9600/R9700S/R9600D: [AMD PRO 26.Q3](https://www.amd.com/en/resources/support-articles/release-notes/RN-PRO-WIN-26-Q3.html), updated August 17,2026, explicitly lists them. Exact first launch remains null. Reference artwork's purchase disclaimer is not interpreted as proof the product itself is unreleased.\n\nServer-only V620/V710, older W5000/WX/FirePro/Vega products, gaming RX GPUs, Instinct and integrated graphics are excluded. All replacement characters were removed from names/evidence. Date-qualified products should be visibly distinct from those with a verified dated launch window.\n\n## LLM evidence second pass\n\n1. [AMD W7900 LLM article](https://www.amd.com/en/blogs/2024/amd-radeon-pro-gpus-and-rocm-software-for-llm-in.html): read body and RPW-462, downloaded and visually checked the original chart, then compared with the June 2 announcement. The1.38x number is **performance per dollar**, not raw token speed. The chart says Llama3 70B-Q4, ROCm6.0 and vLLM; the footnote says Llama3 70b GPTQ. Body prose says Llama2, so the record retains a model-name conflict flag. Four-card deployment discussed elsewhere is not assigned to this single-card claim. No absolute tok/s is derived from prices.\n2. [ToshLLM W6800X](https://toshllm.com/benchmarks/gpu/amd-radeon-pro-w6800x): opened all 8 canonical run pages. [Mixed Vega/Duo group](https://toshllm.com/benchmarks/gpu/amd-radeon-vega-x4-amd-radeon-pro-w6800x-duo-x2): opened all 6 run pages. The second pass fetched all 14 canonical pages again via direct HTTP; every prefill rate, decode rate and displayed quantization matched,28 numerical metrics checked. Four web-reader retries failed, but their direct HTTP pages remained accessible and matched.\n3. Run IDs are preserved for deduplication. These July 2026 v0.82.x/v0.83.x submissions differ from the raw lane's later issue49 v0.88.x logs. Community submissions and verified-tester badges are not treated as independent benchmark reproduction.\n4. Exact displayed model names and quantizations are copied, including Q4_K_M, Q8_0, UD-Q4_K_M and UD-Q4_K_S. **Checkpoint filenames, repositories and revisions are not exposed** on these public run pages, so they remain null and flagged; no Instruct/Base variant is invented. GPT OSS20B Q4_K_M remains the submitted quantization rather than silently becoming MXFP4.\n5. Run pages explicitly report512 prompt tokens. The current contribution protocol says pp512/tg128, but historical run pages do not print generated token counts. Therefore actual `output_tokens` remains null; protocol128 is preserved only in settings.\n6. GPU participation and partitioning are not printed. Both remain unknown. The mixed label `Vega x4 + W6800X Duo x2` is especially ambiguous: its191.88 GB total may describe enumerated devices rather than physical boards. It is **not** converted to two Duo boards/four chips or a homogeneous array. All models and reported multiplicities remain in settings; board count and participating GPU count remain null.\n\nNo measured LLM values were invented for catalog cards without evidence. Official compatibility pages, theoretical compute specs and estimated bandwidth-derived model speeds do not qualify. Remaining LLM leads, inaccessible sheets, duplicate claims and explicit coverage gaps are retained in the coverage queue. Final schema validation verifies only `llm` records in this expansion, unique IDs and valid source/card references.\n\nThe full ToshLLM registry HTML (355 configurations) was additionally checked for W6900X and W6600X. Neither card name was present. Their direct registry GPU pages were unavailable; no registry measurements or estimates were added for them.\n"
    },
    {
      "title": "Llm Scope Quality",
      "text": "# LLM-only cleanup and model identity review\n\nReviewed 2026-10-11 following the user's clarification. The published dataset, SQLite ZIP, filtered exports, statistics and charts now contain LLM inference, serving and language-model training records only. Generic graphics, gaming, rendering, image/video generation and standalone compute-kernel measurements are excluded at compilation. Earlier research inputs remain historical source material; they are not reintroduced into the public dataset by filtering or export.\n\nThe primary selectors are the source's exact model/checkpoint and exact quantization. A three-bit format is not interchangeable with another three-bit format. Exact checkpoint/file names, parameter counts, MoE naming, fine-tunes, quantization suffixes and source aliases are preserved. Case-only quantization normalization retains its original string. Unresolved served aliases remain explicit unknowns rather than being guessed into a canonical model.\n\nCard identity is a catalog foreign key. The comparison view checks card name, exact reported hardware, variant and per-GPU VRAM as well as model, quantization, software, context and concurrency. A W7800 48GB run cannot silently inherit the 32GB product's configuration. Participating GPU chips are separate from physical board counts, including Apple MPX Duo products and mixed systems.\n\nThe previous LLM review findings remain relevant: corrected Puget token counts; source-internal stale latency prose; ambiguous Vulkan device selection in kyuz0 runs; historical versus corrected ggz14 configurations; failed MTP tool qualification; distinct C1/C8 and model-weight follow-up campaigns; actual prompt lengths versus configured token limits; and CPU expert offload. These flags remain attached to retained rows. Forty-five duplicated raw dp-craft metrics were previously removed with duplicate locators retained.\n\nWhole-corpus validation rejects invalid card/source references and structural errors. Scope checks also run against the SQLite export and browser dataset. Catalog-only entries without LLM evidence are displayed as research gaps, not zero performance. The clean model-first explorer is the basis for further collection; no new generic graphics results are being collected.\n"
    },
    {
      "title": "Model Links Labs Quality",
      "text": "# Model-link verification — lab and community lane\n\nVerified on 2026-10-11. This enrichment covers the published LLM rows originating in `community.json`, `labs.json`, `expansion-labs.json` and `expansion-owner.json`. Legacy non-LLM records are outside this task.\n\n## Coverage and checks\n\n- 997 published metric rows, 22 sources and 75 exact model strings.\n- 92 source-specific mappings: 12 tested-weight mappings, 68 upstream/base references, 9 model-family references and 3 unresolved mappings.\n- After precedence resolution: 159 rows have tested-repository/tag links, 774 have base references, 54 have family references and 10 remain unresolved.\n- All rows resolve once at the highest applicable precedence. No missing rows, equal-priority conflicts, invalid record IDs or invalid link fields.\n- 57 distinct repository destinations checked: 53 Hugging Face repositories via the public model metadata API, 3 Ollama library tags and 1 ModelScope repository. All 53 Hugging Face IDs matched exactly; 15 are gated. ModelScope's file API confirmed the named Q4_K_M file.\n- Availability describes the check date. No current repository SHA or mutable tag was represented as the historically tested revision. No model weight files were downloaded.\n- Only the two assigned enrichment files were written. Original model labels, quantization fields, source assessments and benchmark values were preserved.\n\n## Material resolutions\n\nThe [single-card Radiance thread](https://www.reddit.com/r/LocalLLaMA/comments/1viq0pq/qwen36_27b_35b_on_vllm_single_r9700_gfx1201/) has separate original Avesed and revised Intel AutoRound weights. The author's [archived 35B launcher](https://github.com/zzpanic/qwen3.6-vllm-gfx1201-launchers/blob/master/old_work/startup-qwen3.6-35b-vllm.sh) resolves the shortened Avesed name to `Avesed/Qwen3.6-35B-A3B-INT4-W4A16`. Today's root README describes a newer Qwen3.8 campaign and was not applied to old tables. Served-alias-only follow-ups retain base references where their exact tested weights are not independently stated.\n\nThe [Phoronix Lemonade review](https://www.phoronix.com/review/lemonade-vulkan-rocm) identifies release 2026.39.1. Its [versioned catalog](https://github.com/lemonade-sdk/lemonade/blob/v2026.39.1/src/cpp/resources/server_models.json) resolves all five reported aliases: Qwen3-14B Q4_0; Qwen3.5-4B UD-Q4_K_XL; DeepSeek-R1-0528-Qwen3-8B Q4_1; Qwen3-Coder-Next MXFP4_MOE; and MiniCPM4-8B Q4_K_M from ModelScope. The mapping explains that these are release-default artifacts, with no published benchmark download log, override or immutable weight revision.\n\n[LCZ 1538](https://lcz.me/topic/1538) identifies Unsloth Dynamic Q4/Q6/Q8 variants. A bounded 26-record override links their verified publisher repository; unknown-quant and FP8 measurements remain outside it. [LCZ 1718](https://lcz.me/topic/1718) converts named AMD Quark weights into a local GPTQ v3 artifact. Its AMD link is explicitly source weights before conversion, not the resulting tested files.\n\nThe [Pendakwah detailed review](https://pendakwah.tech/reviews/r9700ai/) establishes Qwen2.5-Instruct for its main GGUF sweep and names three Ollama tags. The tags are linked without inferring a historical digest or backfilling today's quantization. Its training link identifies the stated base model, not an unpublished trained adapter.\n\n## Ambiguities deliberately retained\n\n- The [Pendakwah companion](https://pendakwah.tech/asus-radeon-ai-pro-r9700-full-review-32gb-rdna4-for-local-ai) cross-platform table says only 8B Q4. Repeated speed values or neighboring tables do not establish its checkpoint.\n- [Puget's MLPerf Client v1.0 charts](https://www.pugetsystems.com/labs/articles/2025-professional-gpu-content-creation-roundup/) do not name selected models or provide a run manifest. Supported models in the suite cannot identify the actual selection.\n- The Phoronix vLLM suite aggregate cannot be assigned one model repository.\n- Generic Llama 70B, Gemma 2 9B, DeepSeek Coder V2 Lite and abbreviated launch FP8 labels remain family references where base versus Instruct is unreported.\n- Unsloth Qwen3.6 GGUF and MTP repositories overlap in filenames. A publisher name or UD label alone does not choose between them. The `llmfan46 Q5_K_M` row also omits which modified model/version was used.\n- All 34 posts in the [Level1 TP guide](https://forum.level1techs.com/t/250504) and all 75 in the [launch thread](https://forum.level1techs.com/t/239508) were rechecked. Later repository suggestions do not bind earlier benchmark tables. The GPT-OSS BF16/dtype ambiguity remains explicit.\n- The public old PTS llama.cpp profile inspected did not match the models in the later Phoronix campaigns, so its weight URLs were not assigned to those campaigns.\n\n## Source-run navigation advice\n\nA source group is useful navigation across a sweep; it is not a claim of identical configurations. Prefer published run IDs, result files and explicit campaigns. Preserve these boundaries: initial Avesed versus revised Intel versus later GGUF experiments; BetterBench initial versus revised C1/C8 tables; Puget stock TP2 versus tuned/MTP versus layer split; LCZ initial deployment versus September 7 correction and September 15 GPTQ conversion; and the two W6900X posts. Where the source does not establish run boundaries, label the source-level fallback explicitly.\n\n"
    },
    {
      "title": "Model Links Raw Quality",
      "text": "# Model repository mapping: raw lanes\n\nReviewed 2026-10-11. Files written: model-links-raw.json and this report only. Input scope is the current compiled LLM corpus for source IDs owned by research/raw.json and research/expansion-raw.json, excluding kyuz0* and llama10879, which are assigned to the vendor lane. The excluded legacy non-LLM boxwrench source has no published LLM rows and needs no mapping.\n\n## Coverage\n\n104 source/model mappings cover all 6,084 rows in this lane, representing 80 distinct exact model strings. Every mapping includes its source ID; none creates a broad model-wide override that could accidentally affect another author's differently sourced checkpoint.\n\n| Status | Mappings | Covered metric rows |\n| --- | ---: | ---: |\n| tested_weights |31|932|\n| base_model |70|5,134|\n| model_family |2|8|\n| unresolved |1|10|\n\nThere are 57 unique destination URLs. All emitted Hugging Face repository destinations were verified through public API metadata. Named file links were checked against repository file listings. Known gated Llama repositories are marked gated; metadata availability is not claimed to remove download gating. No weight files were downloaded.\n\n## Evidence and exact mappings\n\n- Explicit namespace/file names in VTE, geerlingguy W7700, Wendi, original W7900 ExLlama/vLLM and hazyumps sources map to their named repositories. The original benchmark generally does not pin the HF revision; reasons retain that limit rather than presenting the current file as cryptographically proven historical bytes.\n- R9V aliases are source-specific. The pinned [IQ4_XS package manifest](https://github.com/Dyluhn/R9V/blob/aeee44ff9e37e973bf435bcd8bb0688e9ef1cc29/packages/models/qwen38-flash-next/ud-iq4-xs--mtp-blockfp8--mmproj-q8/package.json), [Q4_K_XL package manifest](https://github.com/Dyluhn/R9V/blob/aeee44ff9e37e973bf435bcd8bb0688e9ef1cc29/packages/models/qwen38-flash-next/ud-q4-k-xl--mtp-blockfp8--mmproj-q8/package.json) and profile descriptors bind aliases to actual published packages. IQ4_XS links to Dyluhn's frozen bf836f0 revision; Q4_K_XL links to Unsloth's frozen 2c41bd2 revision. The Q4 target and shared MTP/projector provenance remain distinct. All seven target shard sizes and SHA256s matched Hugging Face LFS metadata at those exact revisions. The separately named uncensored bundle is never substituted for the ordinary profile.\n- [Quant Lab's custom model](https://huggingface.co/1337Hero/Qwen3.6-27B-Q8_0-ROCMFPX-GGUF) has artifact SHA256 ff4dbc9093c1df6fd1242294d15eb94c7bfe42ed67f98e9d58b9789ac6912c1b in both the benchmark manifest and HF LFS metadata. Its four mappings use the immutable published file revision e3a5f0d47711aa1c06055e2d940c3270457ac45c. The ordinary Q8_0 control was locally converted; it links only to the official BF16 source pinned in the lab's model-source manifest.\n- NikoCloud's qualification table identifies the full DavidAU Turbo MTP model and the outsourc-e Unleashed source. Their matching MTP IQ4_XS/Q4_K_S and Unleashed UD-Q4_K_M files were verified. Local shortened filenames are retained unchanged in mapping keys. The trohrbaugh heretic Q4_K_S is not present as an exact public conversion in the credited source repo; its link is explicitly the fine-tune base reference, not a substituted Q5 or stock model.\n- RWKV web benchmark names an exact dated .st file, whose download repository is specified by the project's guide. The current file moved under old_models_g1; the link points to that verified archive location. Int8/NF4 are runtime transformations of that file. The generic RWKV7 2.9B F16/Q8_0 issue tables do not bind their bytes to the guide's suggested GGUF files, so those two mappings are family references only.\n- bkvargyas' launch script attributes the Flash-Next GPTQ checkpoint to tcclaviger. The verified GPTQ repo remains distinct from hazyumps' non-GPTQ MXFP4-FP8 repo. The local NVFP4-to-MXFP4 conversion has only an upstream base reference because its source publisher namespace is not established.\n\n## Deliberate non-inferences\n\n- Kai Bennett's logs preserve exact local GGUF filenames and quants, but not a publisher namespace or model hash. They receive verified official Qwen upstream links. The presence of UD in a filename is not proof that a specific Unsloth repository supplied the tested bytes; the MTP-suffixed local model is not silently replaced with the standard file.\n- zedbytes' discussion 21043 measurements are separate from JohnTDI's uploaded model mirror. A different commenter posting files does not prove that zedbytes tested those bytes. The eight source/model mappings therefore use the appropriate Qwen3.5 upstream references, while preserving exact quantization and model strings in the database.\n- Codeberg ggz14/Magic Radiance explicitly rewrite the AMD model's MTP head to FP8; their AMD HF links identify the input checkpoint, not the final locally rewritten bytes. Affinity similarly links to the official DeepSeek source before .aff conversion. Radiance TP3's unspecified uncensored fine-tune links only to its verified Flash-Next ancestor.\n- BoJl4apa's Ollama aliases, revision suffixes, MTP/plain tags and unresolved model blobs remain distinct. Model links are verified upstream references, not invented quant publishers. Gemma, Devstral, Mistral, Muse, GLM and Qwen repositories were individually verified rather than inferred from a search URL.\n- ToshLLM's local 128K filename variants retain only parent-model references; neither a public extended-context conversion nor exact GGUF publisher is asserted. The original W7900 MLC conversion also links to its explicit NousResearch input rather than an invented MLC repository.\n\n## Remaining unresolved identity\n\nexp-geerling-r9700-platforms / Qwen3-30B-A3B-Q4_K_M.gguf covers 10 rows. The issue heading/file label says 30B-A3B, but raw llama-bench metadata describes dense 32B and 32.76B parameters. Exact identity cannot be resolved from the original source. This mapping has links:[] and a specific reason. Providing either an MoE or dense repository would hide the contradiction.\n\n## Final validation\n\nRe-read the current database after writing the mapping: all 6,084 in-scope rows have exactly one source/model match; no missing or overlapping mappings. Validated status/link-kind agreement, required source/model keys, nonempty reasons, HTTPS destinations/evidence URLs and unresolved-empty-link behavior. No in-progress placeholders remain. All exact-file links were validated against the destination listing; seven R9V target hashes and the Quant Lab model hash were independently checked. No old research, application, database or Site files were changed.\n"
    },
    {
      "title": "Model Links Vendor Quality",
      "text": "# Vendor model repository links — quality review\n\nVerified on 2026-10-11 against the compiled database.\n\n## Coverage and output\n\n- **76 source/model/quantization selectors**, covering **2,743 benchmark records**, **61 distinct reported model strings**, and **28 sources**.\n- Scope: every kyuz0 source; llama10879 feature/release; all original vendor.json sources still present in the LLM-only compiled database; every expansion-vendor ToshLLM run; and its AMD W7900 result.\n- **6 tested_weights mappings / 41 records**: one exact pinned R9V package, two benchmark-thread protocol mappings, and three source-reported Ollama tags.\n- **63 base_model mappings / 2,689 records** are upstream references. They do not identify the tested quantized weights.\n- **7 model_family mappings / 13 records** preserve unresolved instruction variants or contradictory source identity.\n- No entry lacks a reference link. This does **not** mean all tested checkpoints are resolved: 70 selectors have only upstream/family references.\n- 77 links point to 48 unique destinations: 44 Hugging Face repository roots, one immutable revision, and three Ollama tags.\n\nThe output uses source-specific selectors, preserving exact existing model strings and quantization strings. No old inputs, compiled database, or application files were edited.\n\n## Source provenance checks\n\n### Kyuz0\n\nRead the source at commit `5f73eaeda707daf7f05413476c91bf49aaee064f`, including [README](https://github.com/kyuz0/amd-r9700-ai-toolboxes/blob/5f73eaeda707daf7f05413476c91bf49aaee064f/README.md), [ordinary benchmark runner](https://github.com/kyuz0/amd-r9700-ai-toolboxes/blob/5f73eaeda707daf7f05413476c91bf49aaee064f/benchmark/run_benchmarks.sh), [MTP runner](https://github.com/kyuz0/amd-r9700-ai-toolboxes/blob/5f73eaeda707daf7f05413476c91bf49aaee064f/benchmark/run_mtp_bench.py), ordinary raw logs, and benchmark HTML/summary metadata.\n\nOrdinary logs retain basenames and GGUF internal model labels. The runner discovers local files without recording their download origin. The README shows an Unsloth Qwen3-Coder BF16 example, but it does not prove that all tested files came from that publisher. Matching UD quantization names or a filename found in a live repository does not establish run provenance. These entries therefore link to verified upstream repositories only.\n\nThe MTP source paths identify Qwen3.6 MTP packages. Hugging Face confirms matching files in Unsloth MTP repositories, but the run does not explicitly identify the namespace or revision. Both MTP mappings remain upstream references. They do not silently equate the MTP package with the ordinary model run.\n\n[R9V documentation](https://github.com/kyuz0/amd-r9700-ai-toolboxes/blob/5f73eaeda707daf7f05413476c91bf49aaee064f/docs/r9v-rocm-10.0.md) explicitly links Dyluhn/Qwen3.8-Flash-Next-R9V-IQ4_XS at revision `bf836f0c20b6c92fcad4226ad3115eb8a19f7582`. Both the root repository and that exact revision were verified through the Hugging Face API. This is a tested package mapping.\n\n### llama.cpp discussion 10879\n\nThe [thread's protocol](https://github.com/ggml-org/llama.cpp/discussions/10879) explicitly downloads TheBloke/Llama-2-7B-GGUF's `llama-2-7b.Q4_0.gguf`. The two assigned submissions use that filename. Hugging Face lists the file and identifies the upstream model as meta-llama/Llama-2-7b-hf. These are protocol-supported tested-repository mappings; neither submission reports a checksum or immutable model revision. Their feature/release distinction stays source-specific.\n\n### AMD and HOSTKEY\n\nRechecked AMD's product footnotes, Qwen3.8 and Muse Glimmer articles, OpenClaw article, and the original locally saved ROCm PDF. AMD does not expose a quantizer/revision for these rows, so links are upstream references.\n\n- Muse Glimmer's official publisher is **meta-models**, confirmed by its [primary model card](https://huggingface.co/meta-models/Muse-Glimmer-30B). Guessed meta-llama namespace candidates failed verification and were not published. AMD's preliminary-model qualification stays explicit.\n- FP8-dynamic labels do not establish that RedHatAI or another publisher supplied the tested quantization.\n- The W7900 article's Llama2 prose versus Llama3 chart/footnote conflict remains visible. Its Llama3 70B link is only a family reference, with instruction variant and GPTQ publisher unresolved.\n- [HOSTKEY](https://hostkey.com/blog/142-benchmarking-the-radeon-ai-pro-r9700-amds-red-team-buys-into-on-device-ai/) names three exact Ollama tags. Their live registry pages were checked. Links identify those reported tags, not immutable historical digests or a guessed Hugging Face quantizer.\n\n### Procyon labels\n\n[UL's official model installation table](https://support.benchmarks.ul.com/support/solutions/articles/44002573025-installing-text-generation-models-manually) identifies Mistral-7B-Instruct-v0.2, Llama-3.1-8B-Instruct, Llama-2-13b-chat-hf and Phi-3.5-mini-instruct. This resolves the abbreviated source labels for repository-reference purposes. Their official upstream repositories were checked. The links are explicitly **not** Procyon's converted ONNX/GGUF packages; benchmark-time conversion revisions remain unknown.\n\n### ToshLLM\n\nRe-fetched all 14 original run pages and their 10 model pages. None exposed a Hugging Face repository link or tested checkpoint revision. The individual source selectors therefore retain model labels and quantizations while pointing only to verified references.\n\n- Generic Llama 3.2 1B, Meta Llama 3.1 8B, Gemma 4 26B-A4B and GLM 4 9B labels do not prove an instruction/chat variant. Family links say so explicitly.\n- THUDM GLM repositories redirect to **zai-org**; the canonical destination was independently verified.\n- GPT OSS 20B's reported Q4_K_M label is preserved. The reference to OpenAI's upstream model does not rewrite it to MXFP4.\n\n## Distinct second verification pass\n\nAfter writing the mapping JSON, fetched metadata for **all 44 unique Hugging Face roots** again and checked each canonical repository identity, public/private status, and gating. Every destination was verified; all were public metadata endpoints. The source-reported R9V revision was separately checked. The three Ollama tag pages were opened directly. Gated Llama/Gemma repositories are marked gated, not unavailable.\n\nMatched selectors back to every assigned compiled record: **2,743/2,743 matched exactly one selector**, with zero missing matches and zero overlaps. JSON parse, LF line endings, and replacement-character checks passed. No search-result URL was used as a model repository.\n\n"
    }
  ],
  "coverage_status_counts": {
    "collected": 57,
    "duplicate": 9,
    "needs_extraction": 20,
    "excluded": 25,
    "no_numeric_data": 81,
    "inaccessible": 2
  },
  "card_counts": {
    "w6800": 20,
    "w6600": 0,
    "w6600m": 0,
    "w6400": 0,
    "w6500m": 0,
    "w6300m": 0,
    "w6300": 0,
    "w7900": 157,
    "w7800": 515,
    "w7500": 2,
    "w7600": 2,
    "w7700": 6,
    "w7400": 0,
    "w7500m": 0,
    "w6800x": 16,
    "w6800x-duo": 119,
    "w6900x": 7,
    "w6600x": 0,
    "r9700": 9073,
    "r9600": 0,
    "r9700s": 0,
    "r9600d": 0
  },
  "cards_without_measurements": [
    "w6600",
    "w6600m",
    "w6400",
    "w6500m",
    "w6300m",
    "w6300",
    "w7400",
    "w7500m",
    "w6600x",
    "r9600",
    "r9700s",
    "r9600d"
  ],
  "scope_excluded_record_count": 584,
  "scope_excluded_categories": {
    "image": 44,
    "video": 36,
    "llm": 2,
    "gaming": 86,
    "compute": 184,
    "rendering": 232
  },
  "run_count": 1272,
  "run_grouping_counts": {
    "source_campaign": 78,
    "reported_campaign": 17,
    "reported_run": 14,
    "result_file": 1163
  },
  "model_link_status_counts": {
    "base_model": 8629,
    "model_family": 75,
    "tested_weights": 1193,
    "unresolved": 20
  },
  "model_repository_linked_records": 9897,
  "unmapped_model_record_ids": []
}