{
  "id": "english-grammar-100-2026-09-11",
  "date": "2026-09-11",
  "status": "complete; provisional diagnostic",
  "models": 20,
  "attempts_per_model": 100,
  "attempts": 2000,
  "task": "Raw English sentence completion; no chat template",
  "pass": "Qwen judges both grammaticality and syntactic completeness as true",
  "judge": "Qwen3.8-27B W4A16 AutoRound",
  "judge_temperature": 0,
  "generation": {
    "temperature": 0.8,
    "top_k": 40,
    "top_p": 1,
    "max_new_subword_tokens": 160,
    "max_new_bytes_for_byte_models": 640,
    "seeds": "20261911–20262010",
    "retries_for_quality": false
  },
  "small_model_precision": "float32, Apple MPS",
  "bielik": {
    "model": "speakleash/Bielik-11B-v3.0-Instruct",
    "precision": "Q4_K_M",
    "parameters": 11168796672,
    "sha256": "c16841621efe93c7c8ebf1b374709a96276f3741e649f83ed90131a7b5ad23a8",
    "backend": "llama.cpp raw completion",
    "cache_disabled_repeatability": "4 of 5 sample texts matched; 1 changed; original scored outputs retained"
  },
  "audit": {
    "sample_size": 100,
    "agreement": 92,
    "disagreement": 6,
    "uncertain": 2,
    "method": "Second LLM review; model names and Qwen verdicts omitted from review sheet; same assistant conducted benchmark; not human gold validation; primary scores unchanged"
  },
  "interval": "95% Wilson; covers prompt sampling, not judge error",
  "limitations": [
    "100 hand-written prefixes, one seed per prefix; no benchmark-ladder rank-transfer calibration",
    "Instruction-tuned Bielik and LaMini received bare prefixes, not recommended chat/instruction wrappers",
    "Polish-trained SlayerLab models are English transfer probes, not Polish quality results",
    "Different tokenizers, byte budgets, precision and backend samplers limit exact cross-model comparability",
    "No composite with published five-task scores; small score differences are not reliable rankings"
  ],
  "unavailable": [
    {
      "model": "MobileLLM-125M",
      "reason": "Official gated checkpoint returned HTTP 403"
    },
    {
      "model": "Cerebras-GPT-111M",
      "reason": "Official repository API returned HTTP 404"
    },
    {
      "model": "Muse2-125M-Base",
      "reason": "Weights downloaded; required custom implementation unavailable"
    },
    {
      "model": "Hymba-125M",
      "reason": "Paper reference; no official 125M checkpoint located"
    }
  ],
  "checkpoints": [
    {
      "model": "EleutherAI/gpt-neo-125m",
      "revision": "21def0189f5705e2521767faed922f1f15e7d7db"
    },
    {
      "model": "EleutherAI/pythia-160m",
      "revision": "50f5173d932e8e61f858120bcb800b97af589f46"
    },
    {
      "model": "EleutherAI/pythia-70m",
      "revision": "a39f36b100fe8a5377810d56c3f4789b9c53ac42"
    },
    {
      "model": "HuggingFaceTB/SmolLM-135M",
      "revision": "1d461723eec654e65efdc40cf49301c89c0c92f4"
    },
    {
      "model": "HuggingFaceTB/SmolLM2-135M",
      "revision": "93efa2f097d58c2a74874c7e644dbc9b0cee75a2"
    },
    {
      "model": "MBZUAI/LaMini-GPT-124M",
      "revision": "5c67c8c03c08e82d6138ce2a1eddf5317fac3a6b"
    },
    {
      "model": "McGill-NLP/TLM-100M",
      "revision": "314c821071b7b4556aac052085b0f0abacfd7221"
    },
    {
      "model": "SlayerLab/GoLLeM-110M-PL-v2",
      "revision": "024a67c07d28c6b1c47e0b85e508bf5c21570ea4"
    },
    {
      "model": "SlayerLab/GoLLeM-110M-PL-v3",
      "revision": "ff16d0421e70fc9e771362210da8e749d802ea43"
    },
    {
      "model": "SlayerLab/GoLLeM-45M-PL",
      "revision": "1b1791fcd67727920db39bcf1e7022ce7a209e12"
    },
    {
      "model": "SlayerLab/bdh-25m-pl",
      "revision": "d8895c2f783b922e99281c035148a7a14d5d4199"
    },
    {
      "model": "SlayerLab/goLLeM-110M-PL-SFT-merged",
      "revision": "a319aedb2b705f72c17823deef509225e59e3b0a"
    },
    {
      "model": "SlayerLab/new-training-run-ph-d7e9fe9b",
      "revision": "022fc00d7a2469f383580adfa109902746fc2561"
    },
    {
      "model": "SlayerLab/pollock-mini-lm-125m",
      "revision": "0d22afece64fc5a28f1a32e3eac7a14bc563e089"
    },
    {
      "model": "SlayerLab/slayer-scratch",
      "revision": "8a273e6fd6e05b56d9d05e15e0de8232f8be3548"
    },
    {
      "model": "distilbert/distilgpt2",
      "revision": "2290a62682d06624634c1f46a6ad5be0f47f38aa"
    },
    {
      "model": "facebook/opt-125m",
      "revision": "27dcfa74d334bc871f3234de431e71c6eeba5dd6"
    },
    {
      "model": "openai-community/gpt2",
      "revision": "607a30d783dfa663caf39e06633721c8d4cfcd7e"
    },
    {
      "model": "speakleash/Bielik-11B-v3.0-Instruct",
      "revision": null
    },
    {
      "model": "state-spaces/mamba-130m-hf",
      "revision": "1e76775f628fbf1350fbe4dbb3d971ba64af25a1"
    }
  ]
}