{
  "benchmark_name": "enterprise-swahili-kw5-lite-v2",
  "run_id": "2ff422c0be23",
  "created_at_utc": "2026-09-04T09:29:50.687505+00:00",
  "seed": 20260904,
  "environment": {
    "run_timestamp_utc": "2026-09-04T09:29:50.687505+00:00",
    "python": "3.13.15",
    "platform": "Linux-6.6.122+-x86_64-with-glibc2.35",
    "torch": "2.11.0+cu128",
    "cuda_runtime": "12.8",
    "cuda_available": true,
    "transformers": "4.57.1",
    "datasets": "3.6.0",
    "accelerate": "1.10.0",
    "safetensors": "0.6.2",
    "sentencepiece": "0.2.0",
    "peft": "0.14.0",
    "matplotlib": "3.9.2",
    "numpy": "2.1.3",
    "pandas": "2.2.3",
    "gpu_name": "Tesla T4",
    "gpu_count": 1,
    "gpu_capability": [
      7,
      5
    ]
  },
  "models": {
    "kw5-lite-base": {
      "id": "regnant-io/kw5-lite-base",
      "family": "swahili-specialist",
      "role": "target-base",
      "declared_license": "apache-2.0",
      "chat_format": null,
      "adapter_of": null
    },
    "kw5-lite-instruct": {
      "id": "regnant-io/kw5-lite-instruct",
      "family": "swahili-specialist",
      "role": "target-instruct",
      "declared_license": "apache-2.0",
      "chat_format": "llama2_inst",
      "adapter_of": "regnant-io/kw5-lite-base"
    },
    "swa-gpt2-100mb": {
      "id": "goldfish-models/swa_latn_100mb",
      "family": "swahili-specialist",
      "role": "baseline",
      "declared_license": "apache-2.0",
      "chat_format": null,
      "adapter_of": null
    },
    "gemma-3-270m": {
      "id": "google/gemma-3-270m",
      "family": "multilingual-compact",
      "role": "baseline",
      "declared_license": "gemma-terms",
      "chat_format": null,
      "adapter_of": null
    },
    "qwen2.5-0.5b": {
      "id": "Qwen/Qwen2.5-0.5B",
      "family": "multilingual-compact",
      "role": "baseline",
      "declared_license": "apache-2.0",
      "chat_format": null,
      "adapter_of": null
    },
    "bloom-560m": {
      "id": "bigscience/bloom-560m",
      "family": "multilingual-compact",
      "role": "baseline",
      "declared_license": "bigscience-bloom-rail-1.0",
      "chat_format": null,
      "adapter_of": null
    },
    "xglm-564m": {
      "id": "facebook/xglm-564M",
      "family": "multilingual-compact",
      "role": "baseline",
      "declared_license": "mit",
      "chat_format": null,
      "adapter_of": null
    },
    "pythia-410m": {
      "id": "EleutherAI/pythia-410m",
      "family": "english-only-negative-control",
      "role": "baseline",
      "declared_license": "apache-2.0",
      "chat_format": null,
      "adapter_of": null
    },
    "gpt2-124m": {
      "id": "openai-community/gpt2",
      "family": "english-only-negative-control",
      "role": "baseline",
      "declared_license": "mit",
      "chat_format": null,
      "adapter_of": null
    }
  },
  "generation": {
    "temperature": 0.2,
    "repetition_penalty": 1.3,
    "top_p": 0.95,
    "max_new_tokens": 96,
    "do_sample": true
  },
  "generation_seeds": [
    17,
    23,
    42
  ],
  "kw5_sensitivity": [
    {
      "temperature": 0.1,
      "repetition_penalty": 1.3
    },
    {
      "temperature": 0.2,
      "repetition_penalty": 1.3
    },
    {
      "temperature": 0.2,
      "repetition_penalty": 1.4
    }
  ],
  "collapse_threshold": 0.8,
  "scoring_method": "PMI-normalized suffix likelihood (primary); raw suffix likelihood (secondary, legacy-compatible)",
  "ppl_comparability": "raw PPL retained for reference; bits-per-byte (bpb) is the cross-tokenizer-comparable metric used in the composite score",
  "artifacts": [
    "model_summary.csv",
    "task_rows.csv",
    "generation_outputs.csv",
    "kw5_generation_sensitivity.csv",
    "leaderboard.csv",
    "accuracy_confidence_intervals.csv",
    "kw5_pairwise_bootstrap.csv",
    "model_preflight.csv",
    "label_collapse_audit.csv"
  ],
  "notes": [
    "Primary leaderboard is base-model evaluation, not instruction following.",
    "kw5-lite-instruct is reported separately; not included in base-model pairwise bootstrap.",
    "A gated or failed model remains visible as non-complete rather than being silently replaced or assigned zero.",
    "A model/task pair with >80% single-option prediction share is flagged label_collapsed and excluded from the composite score."
  ]
}