{
 "lastUpdated": "2026-05-21",
 "dataQualityNote": "All results are sourced from AlphaXiv leaderboards or published papers. Speed benchmarks require independent verification. OCR/document rows refreshed from Papers With Code public API dump on 2026-05-21.",
 "results": [
  {
   "schema_version": "codesota-result-v2",
   "result_id": "tts-hardtext-v2:gradium-audrey:2026-05-17",
   "task_id": "text-to-speech",
   "benchmark_id": "codesota-tts-hardtext-v2",
   "model_id": "gradium/tts",
   "model": "Gradium TTS",
   "modelId": "gradium/tts",
   "dataset": "codesota-tts-hardtext-v2",
   "datasetId": "codesota-tts-hardtext-v2",
   "metric": "critical_entity_accuracy",
   "value": 0.7333,
   "source": "codesota-run",
   "sourceUrl": "https://www.codesota.com/text-to-speech/leaderboard",
   "accessDate": "2026-05-17",
   "verified": true,
   "verifiedBy": "CodeSOTA Eval v2",
   "notes": "Compatibility fields mirror the v2 metrics object.",
   "voice": "audrey",
   "checkpoint_or_api_version": null,
   "source_type": "codesota-run",
   "verification_tier": "codesota_measured",
   "date_run": "2026-05-17",
   "sample_count": 30,
   "eval_set_hash": "sha256:hardtext-v2-en-30-prompts",
   "hardware": "api-eu",
   "inference_engine": "api",
   "streaming": true,
   "metrics": {
    "utmos_mean": 4.412,
    "utmos_ci95": [
     null,
     null
    ],
    "wer": 0.1344,
    "cer": 0.067,
    "critical_entity_accuracy": 0.7333,
    "severe_error_count": 6,
    "speaker_similarity_wavlm": null,
    "ttfb_p50_ms": 205.65,
    "ttfb_p95_ms": 299.32,
    "rtf": null,
    "cost_per_1m_chars_usd": 47800
   },
   "artifacts": {
    "prompt_manifest": "/data/tts-intelligibility/runs/gradium-audrey-v1/prompts.json",
    "audio_manifest": "/data/tts-intelligibility/runs/gradium-audrey-v1/audio_manifest.json",
    "audio_hashes": [
     "sha256:gradium-audrey-v1-manifest"
    ],
    "asr_transcripts": "/data/tts-intelligibility/runs/gradium-audrey-v1/asr_transcripts.json",
    "latency_log": "/data/tts-intelligibility/runs/gradium-audrey-v1/latency.json",
    "run_config": "/data/tts-intelligibility/runs/gradium-audrey-v1/run_config.json"
   }
  },
  {
   "schema_version": "codesota-result-v2",
   "result_id": "tts-hardtext-v2:kokoro-af-heart:2026-05-17",
   "task_id": "text-to-speech",
   "benchmark_id": "codesota-tts-hardtext-v2",
   "model_id": "hexgrad/kokoro-82m",
   "model": "Kokoro v1.0",
   "modelId": "hexgrad/kokoro-82m",
   "dataset": "codesota-tts-hardtext-v2",
   "datasetId": "codesota-tts-hardtext-v2",
   "metric": "critical_entity_accuracy",
   "value": 0.6667,
   "source": "codesota-run",
   "sourceUrl": "https://www.codesota.com/text-to-speech/leaderboard",
   "accessDate": "2026-05-17",
   "verified": true,
   "verifiedBy": "CodeSOTA Eval v2",
   "notes": "Compatibility fields mirror the v2 metrics object.",
   "voice": "af_heart",
   "checkpoint_or_api_version": null,
   "source_type": "codesota-run",
   "verification_tier": "codesota_measured",
   "date_run": "2026-05-17",
   "sample_count": 30,
   "eval_set_hash": "sha256:hardtext-v2-en-30-prompts",
   "hardware": "m2-max",
   "inference_engine": "onnxruntime",
   "streaming": true,
   "metrics": {
    "utmos_mean": 4.48,
    "utmos_ci95": [
     null,
     null
    ],
    "wer": 0.1557,
    "cer": 0.0675,
    "critical_entity_accuracy": 0.6667,
    "severe_error_count": 6,
    "speaker_similarity_wavlm": null,
    "ttfb_p50_ms": 855.02,
    "ttfb_p95_ms": 2123.32,
    "rtf": null,
    "cost_per_1m_chars_usd": 0
   },
   "artifacts": {
    "prompt_manifest": "/data/tts-intelligibility/runs/kokoro-af-heart-v1/prompts.json",
    "audio_manifest": "/data/tts-intelligibility/runs/kokoro-af-heart-v1/audio_manifest.json",
    "audio_hashes": [
     "sha256:kokoro-af-heart-v1-manifest"
    ],
    "asr_transcripts": "/data/tts-intelligibility/runs/kokoro-af-heart-v1/asr_transcripts.json",
    "latency_log": "/data/tts-intelligibility/runs/kokoro-af-heart-v1/latency.json",
    "run_config": "/data/tts-intelligibility/runs/kokoro-af-heart-v1/run_config.json"
   }
  },
  {
   "model": "Peng et al. 2023 (WRN-70-16)",
   "modelId": "",
   "dataset": "robustbench-cifar10-linf",
   "datasetId": "robustbench-cifar10-linf",
   "metric": "Robust Accuracy",
   "value": 71.07,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//robustbench-cifar10-linf",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Wang et al. 2023 (WRN-70-16)",
   "modelId": "",
   "dataset": "robustbench-cifar10-linf",
   "datasetId": "robustbench-cifar10-linf",
   "metric": "Robust Accuracy",
   "value": 70.69,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//robustbench-cifar10-linf",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gowal et al. 2021 (WRN-70-16)",
   "modelId": "",
   "dataset": "robustbench-cifar10-linf",
   "datasetId": "robustbench-cifar10-linf",
   "metric": "Robust Accuracy",
   "value": 66.11,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//robustbench-cifar10-linf",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Grounding DINO 1.5 Pro",
   "modelId": "",
   "dataset": "lvis-zero-shot",
   "datasetId": "lvis-zero-shot",
   "metric": "ap",
   "value": 47.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//lvis-zero-shot",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OWLv2 (ViT-L)",
   "modelId": "",
   "dataset": "lvis-zero-shot",
   "datasetId": "lvis-zero-shot",
   "metric": "ap",
   "value": 44.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//lvis-zero-shot",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "YOLO-World v2-X",
   "modelId": "",
   "dataset": "lvis-zero-shot",
   "datasetId": "lvis-zero-shot",
   "metric": "ap",
   "value": 35.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//lvis-zero-shot",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CodeLlama-34B",
   "modelId": "",
   "dataset": "apps",
   "datasetId": "apps",
   "metric": "pass@5",
   "value": 32.81,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//apps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CodeLlama-13B",
   "modelId": "",
   "dataset": "apps",
   "datasetId": "apps",
   "metric": "pass@5",
   "value": 23.74,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//apps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CodeLlama-7B",
   "modelId": "",
   "dataset": "apps",
   "datasetId": "apps",
   "metric": "pass@5",
   "value": 10.76,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//apps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Ultravox-GLM-4P7",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 88.86,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Whisper-v3-large + GPT-4o (cascade)",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 87.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o-Audio",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 86.75,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Whisper-v3-large + LLaMA-3.1-8B (cascade)",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 77.48,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Kimi-Audio",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 76.91,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniCPM-o",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 71.23,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VITA-1.5",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 64.53,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-Audio",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 55.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Omni",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 41.12,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VITA-1.0",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 36.43,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Mini-Omni2",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 33.49,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Mini-Omni",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 30.42,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Moshi",
   "modelId": "",
   "dataset": "voicebench",
   "datasetId": "voicebench",
   "metric": "overall-score",
   "value": 29.51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//voicebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-VL 72B",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 85.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-VL 72B",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 84.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL2-76B",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 84.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 3.2 Vision 90B",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 83.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 1.5 Pro",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 82.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4V",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 78,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 77.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-1.5",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 61.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "textvqa",
   "datasetId": "textvqa",
   "metric": "accuracy",
   "value": 42.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//textvqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-VL-72B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 63.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "seed-1.6-vision",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 62.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gemini-25-pro",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 62.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-VL-72B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 61.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "qwen3-omni-30b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 61.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "nemotron-nano-v2-vl",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 61.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 60.77,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gemini-25-pro",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 59.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "minicpm-v-4.5-8b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 58.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "sail-vl2-8b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 57.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "llama-3.1-nemotron-nano-vl-8b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 56.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 56,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL3-14B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 55.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-VL-7B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 55.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 55.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ovis2.5-8b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 54.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL3-14B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 52.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 1.5 Pro",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 51.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gemini-1.5-pro",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 51.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "sail-vl2-8b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 49.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Ovis2-8B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 49.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "claude-3.5-sonnet",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 48.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "minicpm-v-4.5-8b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 48.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-VL-72B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 47.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Ovis2-8B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 47.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o-2024",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 47.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "claude-3.5-sonnet",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 47.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "internvl3.5-14b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 47.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "step-1v",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 46.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-VL-7B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 46.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Step-1V",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 46.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 46.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL2.5-78B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 46.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-VL-72B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 46.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o-2024",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 45.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Claude 3.5 Sonnet",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 45.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniCPM-o-2.6",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 45.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL2.5-78B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 45,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "grok4",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 45,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o-mini",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 44.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-VL2-Small",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 43.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 1.5 Pro",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 43.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-VL2-Small",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 42.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GLM-4V-9B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 42.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Step-1V",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 42.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "claude-sonnet-4",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 42.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "qwen2.5-vl-7b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 41.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniCPM-o-2.6",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 41.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "deepseek-vl2-small",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 41,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Pixtral-12B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 40.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Claude 3.5 Sonnet",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 39.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "pixtral-12b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 38.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "phi-4-multimodal",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 38.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "glm-4v-9b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 37.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GLM-4V-9B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 36.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-OneVision-7B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 36.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cambrian-1-8B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 34.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Molmo-7B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 34.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "molmo-7b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 33.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "llava-ov-7b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 33.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 32.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-NeXT-8B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 31.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "idefics3-8b",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 26,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-2512",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 25.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TextMonkey",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 23.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "docowl2",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 23.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Monkey",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-public",
   "value": 23.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-OneVision-7B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 17.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TextMonkey",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 15.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Pixtral-12B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 14.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Monkey",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 13.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Molmo-7B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 12.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cambrian-1-8B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 9.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-NeXT-8B",
   "modelId": "",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-public",
   "value": 9.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ocrbench-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NV-Embed-v2",
   "modelId": "",
   "dataset": "beir",
   "datasetId": "beir",
   "metric": "ndcg@10",
   "value": 62.65,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//beir",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GTE-Qwen2-7B-instruct",
   "modelId": "",
   "dataset": "beir",
   "datasetId": "beir",
   "metric": "ndcg@10",
   "value": 60.25,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//beir",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "E5-Mistral-7B-instruct",
   "modelId": "",
   "dataset": "beir",
   "datasetId": "beir",
   "metric": "ndcg@10",
   "value": 56.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//beir",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ColBERTv2",
   "modelId": "",
   "dataset": "beir",
   "datasetId": "beir",
   "metric": "ndcg@10",
   "value": 49.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//beir",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "RankLLaMA-7B",
   "modelId": "",
   "dataset": "ms-marco",
   "datasetId": "ms-marco",
   "metric": "mrr@10",
   "value": 41.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ms-marco",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "jina-reranker-v2-base-multilingual",
   "modelId": "",
   "dataset": "ms-marco",
   "datasetId": "ms-marco",
   "metric": "mrr@10",
   "value": 41.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ms-marco",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ColBERTv2",
   "modelId": "",
   "dataset": "ms-marco",
   "datasetId": "ms-marco",
   "metric": "mrr@10",
   "value": 39.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ms-marco",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MonoT5-3B",
   "modelId": "",
   "dataset": "ms-marco",
   "datasetId": "ms-marco",
   "metric": "mrr@10",
   "value": 39,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ms-marco",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NV-Embed-v2",
   "modelId": "",
   "dataset": "mteb",
   "datasetId": "mteb",
   "metric": "avg-score",
   "value": 72.31,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mteb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GTE-Qwen2-7B-instruct",
   "modelId": "",
   "dataset": "mteb",
   "datasetId": "mteb",
   "metric": "avg-score",
   "value": 72.05,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mteb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "voyage-3-large",
   "modelId": "",
   "dataset": "mteb",
   "datasetId": "mteb",
   "metric": "avg-score",
   "value": 70.32,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mteb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "E5-Mistral-7B-instruct",
   "modelId": "",
   "dataset": "mteb",
   "datasetId": "mteb",
   "metric": "avg-score",
   "value": 66.63,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mteb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "jina-embeddings-v3",
   "modelId": "",
   "dataset": "mteb",
   "datasetId": "mteb",
   "metric": "avg-score",
   "value": 65.18,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mteb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "text-embedding-3-large",
   "modelId": "",
   "dataset": "mteb",
   "datasetId": "mteb",
   "metric": "avg-score",
   "value": 64.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mteb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GTE-Qwen2-7B-instruct",
   "modelId": "",
   "dataset": "sts-benchmark",
   "datasetId": "sts-benchmark",
   "metric": "spearman",
   "value": 88.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//sts-benchmark",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "E5-Mistral-7B-instruct",
   "modelId": "",
   "dataset": "sts-benchmark",
   "datasetId": "sts-benchmark",
   "metric": "spearman",
   "value": 84.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//sts-benchmark",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "all-MiniLM-L6-v2",
   "modelId": "",
   "dataset": "sts-benchmark",
   "datasetId": "sts-benchmark",
   "metric": "spearman",
   "value": 82.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//sts-benchmark",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "bestfitting (1st place ensemble)",
   "modelId": "",
   "dataset": "severstal-steel-defect",
   "datasetId": "severstal-steel-defect",
   "metric": "Dice",
   "value": 0.90883,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//severstal-steel-defect",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "2nd Place Solution",
   "modelId": "",
   "dataset": "severstal-steel-defect",
   "datasetId": "severstal-steel-defect",
   "metric": "Dice",
   "value": 0.9084,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//severstal-steel-defect",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "U-Net Ensemble (Pavlov)",
   "modelId": "",
   "dataset": "severstal-steel-defect",
   "datasetId": "severstal-steel-defect",
   "metric": "Dice",
   "value": 0.903,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//severstal-steel-defect",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Kling 1.0",
   "modelId": "",
   "dataset": "vbench",
   "datasetId": "vbench",
   "metric": "total-score",
   "value": 85.37,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Runway Gen-3 Alpha",
   "modelId": "",
   "dataset": "vbench",
   "datasetId": "vbench",
   "metric": "total-score",
   "value": 85.22,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CogVideoX-5B",
   "modelId": "",
   "dataset": "vbench",
   "datasetId": "vbench",
   "metric": "total-score",
   "value": 82.75,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Open-Sora 1.2",
   "modelId": "",
   "dataset": "vbench",
   "datasetId": "vbench",
   "metric": "total-score",
   "value": 80.91,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BiGTex",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-products",
   "value": 90.29,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BiGTex",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 88.51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GLEM+GIANT+SAGN+SCR",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-products",
   "value": 87.37,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LD+GIANT+SAGN+SCR",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-products",
   "value": 87.18,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GraDBERT & RevGAT+KD",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-products",
   "value": 86.92,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GraphSAGE",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-products",
   "value": 83.89,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GCN",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-products",
   "value": 82.33,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GAT",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-products",
   "value": 80.99,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SimTeG+TAPE+RevGAT",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 78.03,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TAPE+RevGAT",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 77.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SimTeG+TAPE+GraphSAGE",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 77.48,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LD+REVGAT",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 77.26,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GraDBERT & RevGAT+KD",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 77.21,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GLEM+RevGAT",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 76.94,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GCN",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 73.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GAT",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 73.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GraphSAGE",
   "modelId": "",
   "dataset": "ogb",
   "datasetId": "ogb",
   "metric": "accuracy-ogbn-arxiv",
   "value": 72.95,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ogb",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 53.65,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 52.93,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 50.28,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 49.33,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 48.68,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 48.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 48.55,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 47.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 44.93,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 44.93,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 44.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 44.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 44.03,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 43.85,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 43.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 42.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 42.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 41.67,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 41.63,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 41.53,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multi-image-reasoning",
   "value": 39.65,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 39.28,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 39.23,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 38.14,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 37.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 37.22,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 36.37,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 36.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 33.58,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 33.53,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 33.27,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 33,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 32.47,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 32,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 32,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 30.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "grounded-qa",
   "value": 30.27,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 28.33,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 27.73,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 27.15,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 27.15,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 26.92,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-13B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 26.59,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "knowledge-images-qa",
   "value": 26.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 26.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 26.12,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 25.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 25.83,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 25.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 25.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 24.76,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 24.51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 24.41,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 24.22,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 23.13,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (LLaMA2-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 22.95,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 22.21,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Cheetah (Vicuna-7B)",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 22.15,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 21.65,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 21.31,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "accuracy",
   "value": 21.24,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 21.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 19.33,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 18,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 17.94,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 17.57,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 17.07,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 16.88,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 16.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 16.25,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 16,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "relation-cloze",
   "value": 15.85,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 15.57,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 15.37,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 14.22,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenFlamingo",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 13.85,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 13.69,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaMA-Adapter V2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 13.51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 12.67,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructBLIP",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 11.49,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Otter",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 11.39,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "storytelling",
   "value": 10.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 10.67,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 8.27,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiniGPT-4",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 7.95,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "multimodal-dialogue",
   "value": 7.79,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mPLUG-Owl",
   "modelId": "",
   "dataset": "demon-bench",
   "datasetId": "demon-bench",
   "metric": "visual-inference",
   "value": 5.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//demon-bench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InterFuser",
   "modelId": "",
   "dataset": "carla-leaderboard",
   "datasetId": "carla-leaderboard",
   "metric": "driving_score",
   "value": 76.18,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//carla-leaderboard",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TCP",
   "modelId": "",
   "dataset": "carla-leaderboard",
   "datasetId": "carla-leaderboard",
   "metric": "driving_score",
   "value": 75.14,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//carla-leaderboard",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Think2Drive",
   "modelId": "",
   "dataset": "carla-leaderboard",
   "datasetId": "carla-leaderboard",
   "metric": "driving_score",
   "value": 46,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//carla-leaderboard",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mineru-2.5",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "layout-map",
   "value": 97.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PaddleOCR-VL-1.5",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 94.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "paddleocr-vl",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "table-teds",
   "value": 93.52,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "paddleocr-vl",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 92.86,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "paddleocr-vl-0.9b",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 92.56,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "formula-cdm",
   "value": 92.43,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "reading-order",
   "value": 91.63,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "table-teds",
   "value": 91.02,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 3 Pro",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 90.33,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Dolphin-v2",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 89.78,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "qwen3-vl-235b",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 89.15,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ocrverse-4b",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 88.56,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "dots-ocr-3b",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 88.41,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gemini-25-pro",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 88.03,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "qwen25-vl",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 87.02,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PP-StructureV3",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 86.73,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-OCR",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 86.46,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "clearocr-teamquest",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "reading-order",
   "value": 86.04,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Nanonets-OCR-s",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 85.59,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MinerU2-VLM",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 85.56,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Dolphin-1.5",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 85.06,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL3.5-241B",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 82.67,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmOCR-7B",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 81.79,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "POINTS-Reader",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 80.98,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL3-76B",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 80.33,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 79.75,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-2512",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 79.75,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MinerU2-pipeline",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 75.51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 75.02,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OCRFlux-3B",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 74.82,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Dolphin",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 74.67,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Marker 1.8.2",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 71.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "table-teds",
   "value": 70.88,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "clearocr-teamquest",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 31.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "clearocr-teamquest",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "formula-edit-distance",
   "value": 0.902,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "clearocr-teamquest",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "table-teds",
   "value": 0.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "formula-edit-distance",
   "value": 0.218,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "clearocr-teamquest",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.154,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.099,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit",
   "value": 0.041,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o",
   "modelId": "",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "ocr-edit-distance",
   "value": 0.02,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//omnidocbench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "RVT-2",
   "modelId": "",
   "dataset": "rlbench",
   "datasetId": "rlbench",
   "metric": "success-rate",
   "value": 81.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rlbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "RVT",
   "modelId": "",
   "dataset": "rlbench",
   "datasetId": "rlbench",
   "metric": "success-rate",
   "value": 62.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rlbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PerAct",
   "modelId": "",
   "dataset": "rlbench",
   "datasetId": "rlbench",
   "metric": "success-rate",
   "value": 43.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rlbench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OVRL-V2",
   "modelId": "",
   "dataset": "habitat-objectnav-hm3d",
   "datasetId": "habitat-objectnav-hm3d",
   "metric": "success_rate",
   "value": 64.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//habitat-objectnav-hm3d",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Habitat-Web",
   "modelId": "",
   "dataset": "habitat-objectnav-hm3d",
   "datasetId": "habitat-objectnav-hm3d",
   "metric": "success_rate",
   "value": 35.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//habitat-objectnav-hm3d",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "WavLLM",
   "modelId": "",
   "dataset": "audiobench",
   "datasetId": "audiobench",
   "metric": "avg-score",
   "value": 50.25,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiobench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SALMONN",
   "modelId": "",
   "dataset": "audiobench",
   "datasetId": "audiobench",
   "metric": "avg-score",
   "value": 43.99,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiobench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-Audio-Instruct",
   "modelId": "",
   "dataset": "audiobench",
   "datasetId": "audiobench",
   "metric": "avg-score",
   "value": 42.12,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiobench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Whisper+LLaMA-3 (cascade)",
   "modelId": "",
   "dataset": "audiobench",
   "datasetId": "audiobench",
   "metric": "avg-score",
   "value": 40.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiobench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen-Audio-Chat",
   "modelId": "",
   "dataset": "audiobench",
   "datasetId": "audiobench",
   "metric": "avg-score",
   "value": 38.59,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiobench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BEATs",
   "modelId": "",
   "dataset": "audioset",
   "datasetId": "audioset",
   "metric": "map",
   "value": 0.506,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audioset",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AST",
   "modelId": "",
   "dataset": "audioset",
   "datasetId": "audioset",
   "metric": "map",
   "value": 0.485,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audioset",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HTS-AT",
   "modelId": "",
   "dataset": "audioset",
   "datasetId": "audioset",
   "metric": "map",
   "value": 0.471,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audioset",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CLAP",
   "modelId": "",
   "dataset": "audioset",
   "datasetId": "audioset",
   "metric": "map",
   "value": 0.428,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audioset",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TD3",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 5592,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SAC",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 5179,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PPO",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 2038,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TD-MPC2 (317M params)",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 960,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TD-MPC2 (19M params)",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 953,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "FOWM",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 945,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BRO",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 941,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TD-MPC2 (5M params)",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 929,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DreamerV3",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 897,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TD-MPC",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 857,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DrQ-v2",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 799,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SAC (state-based)",
   "modelId": "",
   "dataset": "mujoco",
   "datasetId": "mujoco",
   "metric": "average-return",
   "value": 777,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mujoco",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "go-explore",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 40000,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "agent57",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 4731.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MEME",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 4087,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "bbos-1",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 1100,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gdi-h3",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 950,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "dreamerv3",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 840,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "muzero",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 731,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EfficientZero V2",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 242.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "rainbow-dqn",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 231,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BBF (Bigger, Better, Faster)",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 224.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DIAMOND",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 145.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "STORM",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 126.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Simulus",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 110,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DART",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 102.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "human-gamer",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 100,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "dqn",
   "modelId": "",
   "dataset": "atari-2600",
   "datasetId": "atari-2600",
   "metric": "human-normalized-score",
   "value": 79,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//atari-2600",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SegFormer-B5",
   "modelId": "",
   "dataset": "cityscapes",
   "datasetId": "cityscapes",
   "metric": "miou",
   "value": 84,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cityscapes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Mask2Former (Swin-L)",
   "modelId": "",
   "dataset": "cityscapes",
   "datasetId": "cityscapes",
   "metric": "miou",
   "value": 83.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cityscapes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OneFormer (DiNAT-L)",
   "modelId": "",
   "dataset": "cityscapes",
   "datasetId": "cityscapes",
   "metric": "miou",
   "value": 83,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cityscapes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-VL 72B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 87.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVL2-76B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 87.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 1.5 Pro",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 86.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PaLI-X 55B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 86.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NVLM-D 1.0 72B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 85.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NVLM-X 1.0 72B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 85.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NVLM-H 1.0 72B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 85.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VILA-1.5 40B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 84.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-NeXT 34B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 83.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-NeXT 13B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 82.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CogVLM-17B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 82.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-NeXT 7B (Mistral)",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 82.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 82.19,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-NeXT 7B (Vicuna)",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 81.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Pixtral Large",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 80.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 3-V 405B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 80.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-1.5 13B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 80,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaVA-1.5",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 80,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 3-V 70B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 79.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Pixtral-12B",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 78.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 78.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 3.2 90B Vision Instruct",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 78.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4V",
   "modelId": "",
   "dataset": "vqa-v2",
   "datasetId": "vqa-v2",
   "metric": "accuracy",
   "value": 77.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vqa-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "base",
   "value": 99.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "base",
   "value": 99.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LightOnOCR-2-1B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "base",
   "value": 99.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "base",
   "value": 99.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "headers-footers",
   "value": 96.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.3.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "headers-footers",
   "value": 95.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "long-tiny-text",
   "value": 92.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "multi-column",
   "value": 92.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LightOnOCR-2-1B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "long-tiny-text",
   "value": 91.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "headers-footers",
   "value": 90.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LightOnOCR-2-1B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "arxiv",
   "value": 89.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LightOnOCR-2-1B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "tables",
   "value": 89,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "dots-ocr-3b",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "tables",
   "value": 88.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "tables",
   "value": 88,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LightOnOCR-2-1B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans-math",
   "value": 85.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "tables",
   "value": 84.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LightOnOCR-2-1B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "multi-column",
   "value": 84.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "marker-1.10.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "arxiv",
   "value": 83.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "multi-column",
   "value": 83.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 83.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "arxiv",
   "value": 83,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 82.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans-math",
   "value": 82.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "arxiv",
   "value": 82.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "long-tiny-text",
   "value": 81.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "tables",
   "value": 81.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "multi-column",
   "value": 81.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "long-tiny-text",
   "value": 80.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans-math",
   "value": 80.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "arxiv",
   "value": 80.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.3.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans-math",
   "value": 79.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen3-VL-4B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 79.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PaddleOCR-VL-1.5",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 79.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "dots-ocr-3b",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 79.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 78,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "marker-1.10.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 76.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "marker-1.10.1",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 76.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MonkeyOCR-pro-3B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 75.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "deepseek-ocr",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 75.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mineru-2.5",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 75.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans",
   "value": 73.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mistral-ocr-api",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 72,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o-anchored",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 69.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "nanonets-ocr2-3b",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 69.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gemini-flash-2",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 63.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "chandra-ocr-0.1.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans",
   "value": 50.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "olmocr-v0.4.0",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans",
   "value": 47.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LightOnOCR-2-1B",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans",
   "value": 42.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qianfan-OCR",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "headers-footers",
   "value": 42,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o",
   "modelId": "",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "old-scans",
   "value": 40.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//olmocr-bench",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Med-Gemini",
   "modelId": "",
   "dataset": "medqa-usmle",
   "datasetId": "medqa-usmle",
   "metric": "Accuracy",
   "value": 91.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//medqa-usmle",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Med-PaLM 2",
   "modelId": "",
   "dataset": "medqa-usmle",
   "datasetId": "medqa-usmle",
   "metric": "Accuracy",
   "value": 86.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//medqa-usmle",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4 (base)",
   "modelId": "",
   "dataset": "medqa-usmle",
   "datasetId": "medqa-usmle",
   "metric": "Accuracy",
   "value": 86.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//medqa-usmle",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LayoutLMv3-large",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 92.08,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "UDOP",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 91.62,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LayoutLMv3-base",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 90.29,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DocFormerv2-large",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 88.89,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LiLT[EN-R2]-base",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 88.41,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DocFormerv2-base",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 88.37,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "StructuralLM",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 85.14,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "FormNet",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 84.69,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BROS-large",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 84.52,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LayoutLMv2-large",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 84.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LayoutLMv2-base",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 82.76,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LayoutLMv1-base",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 79.27,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LayoutLMv1-large",
   "modelId": "",
   "dataset": "funsd",
   "datasetId": "funsd",
   "metric": "f1",
   "value": 77.89,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//funsd",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-R1-0528",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 73.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen3-235B-A22B",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 70.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-R1",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 65.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-R1-Distill-Llama-70B",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 65.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenAI o1 (Dec 2024)",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 63.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Kimi k1.5 (long-CoT)",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 62.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-R1-Distill-Qwen-32B",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 62.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-R1-Distill-Qwen-14B",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 59.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o1-mini",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 53.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-V3-0324",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 49.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-R1-Distill-Qwen-7B",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 49.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-R1-Distill-Llama-8B",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 49,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Kimi k1.5 (short-CoT)",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 47.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 4 Maverick (17B-128E)",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 43.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-V3",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 40.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 27B IT",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 39,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Claude 3.5 Sonnet",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 38.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 32.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 4 Scout (17B-16E)",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 32.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 12B IT",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 32,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-Coder-32B-Instruct",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 31.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 4B IT",
   "modelId": "",
   "dataset": "livecodebench",
   "datasetId": "livecodebench",
   "metric": "pass@1",
   "value": 23,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//livecodebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o4-mini (high)",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 99.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o3-mini (high)",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 97.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o4-mini",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 97.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o3-mini",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 96.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-41",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 94.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4.1 mini",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 93.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-Coder-32B-Instruct",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 92.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o1-preview",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 92.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o1-mini",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 92.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Claude 3.5 Sonnet (Oct 2024)",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 92.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 92,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 91,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o (Nov 2024)",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 90.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "llama-31-405b",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 89,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-45-preview",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 88.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "grok-2",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 88.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-Coder-7B-Instruct",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 88.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o3 (high)",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 88.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4-turbo",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 88.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 27B IT",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 87.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "o3",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 87.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o-mini",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 87.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4.1 nano",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 87,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 12B IT",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 85.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-Coder-V2-Instruct",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 85.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "claude-3-opus",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 84.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Phi-4 (14B)",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 82.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "deepseek-v3",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 82.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "llama-3-70b",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 81.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "llama-31-70b",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 80.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 71.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 4B IT",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 71.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-V3",
   "modelId": "",
   "dataset": "humaneval",
   "datasetId": "humaneval",
   "metric": "pass@1",
   "value": 65.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DINOv2 ViT-g/14",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top1_accuracy",
   "value": 86.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DINOv2 ViT-g/14",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top-1-accuracy",
   "value": 86.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DINOv2 ViT-L/14",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top-1-accuracy",
   "value": 86.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CLIP ViT-L/14",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top-1-accuracy",
   "value": 85.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SimCLRv2 (ResNet-152 3x)",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top1_accuracy",
   "value": 79.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MAE ViT-H/14",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top-1-accuracy",
   "value": 77.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MAE ViT-H/14",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top1_accuracy",
   "value": 76.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MAE ViT-L/16",
   "modelId": "",
   "dataset": "imagenet-linear-probe",
   "datasetId": "imagenet-linear-probe",
   "metric": "top-1-accuracy",
   "value": 76,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-linear-probe",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Nova 2",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "cer",
   "value": 10.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-Audio",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "cer",
   "value": 9.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Scribe V1",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "cer",
   "value": 8.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Whisper Large V3",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "cer",
   "value": 7.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 2.5 Pro",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "cer",
   "value": 6.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o Transcribe",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "cer",
   "value": 6.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 3 Pro",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "cer",
   "value": 6.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Nova 2",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "wer",
   "value": 6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2-Audio",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "wer",
   "value": 5.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Whisper Large V3",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "wer",
   "value": 4.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 2.5 Pro",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "wer",
   "value": 3.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Scribe V1",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "wer",
   "value": 3.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemini 3 Pro",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "wer",
   "value": 2.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o Transcribe",
   "modelId": "",
   "dataset": "wildasr",
   "datasetId": "wildasr",
   "metric": "wer",
   "value": 2.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wildasr",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AutoAttack vs Undefended ResNet",
   "modelId": "",
   "dataset": "robustbench-cifar10-linf-attack",
   "datasetId": "robustbench-cifar10-linf-attack",
   "metric": "Attack Success Rate",
   "value": 100,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//robustbench-cifar10-linf-attack",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AutoAttack vs Wang 2023",
   "modelId": "",
   "dataset": "robustbench-cifar10-linf-attack",
   "datasetId": "robustbench-cifar10-linf-attack",
   "metric": "Attack Success Rate",
   "value": 29.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//robustbench-cifar10-linf-attack",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AutoAttack vs Peng 2023",
   "modelId": "",
   "dataset": "robustbench-cifar10-linf-attack",
   "datasetId": "robustbench-cifar10-linf-attack",
   "metric": "Attack Success Rate",
   "value": 28.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//robustbench-cifar10-linf-attack",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TAPE + RevGAT",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 92.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AuGLM (T5-large)",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 91.51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ENGINE",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 91.48,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InstructGLM",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 90.77,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GLEM + RevGAT",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 88.56,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GCNLLMEmb",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 88.15,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LLaGA (Mistral-7B)",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 87.55,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SDGAT",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 85.29,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GCN* (tuned)",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 85.08,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GAT* (tuned)",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 84.64,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SGFormer",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 84.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GraphSAGE* (tuned)",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 84.18,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Polynormer",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 83.25,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GOAT",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 83.18,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GAT",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 83,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GraphGPS",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 82.84,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Exphormer",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 82.77,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GraphSAGE",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 82.68,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NodeFormer",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 82.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NAGphormer",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 82.12,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GCN",
   "modelId": "",
   "dataset": "cora",
   "datasetId": "cora",
   "metric": "accuracy",
   "value": 81.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cora",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViTPose-H",
   "modelId": "",
   "dataset": "coco-keypoints",
   "datasetId": "coco-keypoints",
   "metric": "ap",
   "value": 80.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//coco-keypoints",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "RTMPose-X",
   "modelId": "",
   "dataset": "coco-keypoints",
   "datasetId": "coco-keypoints",
   "metric": "ap",
   "value": 78.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//coco-keypoints",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HRNet-W48",
   "modelId": "",
   "dataset": "coco-keypoints",
   "datasetId": "coco-keypoints",
   "metric": "ap",
   "value": 75.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//coco-keypoints",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GROVER-Large",
   "modelId": "",
   "dataset": "moleculenet-bbbp",
   "datasetId": "moleculenet-bbbp",
   "metric": "ROC-AUC",
   "value": 0.94,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//moleculenet-bbbp",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "D-MPNN (ChemProp)",
   "modelId": "",
   "dataset": "moleculenet-bbbp",
   "datasetId": "moleculenet-bbbp",
   "metric": "ROC-AUC",
   "value": 0.913,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//moleculenet-bbbp",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MolCLR",
   "modelId": "",
   "dataset": "moleculenet-bbbp",
   "datasetId": "moleculenet-bbbp",
   "metric": "ROC-AUC",
   "value": 0.736,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//moleculenet-bbbp",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ZoeDepth-N",
   "modelId": "",
   "dataset": "nyu-depth-v2",
   "datasetId": "nyu-depth-v2",
   "metric": "absrel",
   "value": 0.075,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nyu-depth-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Marigold",
   "modelId": "",
   "dataset": "nyu-depth-v2",
   "datasetId": "nyu-depth-v2",
   "metric": "absrel",
   "value": 0.055,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nyu-depth-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiDaS 3.1 (BEiT-512)",
   "modelId": "",
   "dataset": "nyu-depth-v2",
   "datasetId": "nyu-depth-v2",
   "metric": "absrel",
   "value": 0.048,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nyu-depth-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Depth Anything V1 (ViT-L)",
   "modelId": "",
   "dataset": "nyu-depth-v2",
   "datasetId": "nyu-depth-v2",
   "metric": "absrel",
   "value": 0.045,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nyu-depth-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Depth Anything V2 (ViT-L)",
   "modelId": "",
   "dataset": "nyu-depth-v2",
   "datasetId": "nyu-depth-v2",
   "metric": "absrel",
   "value": 0.041,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nyu-depth-v2",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Megatron-BERT",
   "modelId": "",
   "dataset": "race",
   "datasetId": "race",
   "metric": "accuracy",
   "value": 90.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//race",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ALBERT (Ensemble)",
   "modelId": "",
   "dataset": "race",
   "datasetId": "race",
   "metric": "accuracy",
   "value": 89.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//race",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4",
   "modelId": "",
   "dataset": "xnli",
   "datasetId": "xnli",
   "metric": "accuracy",
   "value": 87.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//xnli",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "XLM-RoBERTa-large",
   "modelId": "",
   "dataset": "xnli",
   "datasetId": "xnli",
   "metric": "accuracy",
   "value": 83.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//xnli",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "mDeBERTa-v3-base",
   "modelId": "",
   "dataset": "xnli",
   "datasetId": "xnli",
   "metric": "accuracy",
   "value": 80.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//xnli",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Puigcerver",
   "modelId": "",
   "dataset": "rimes",
   "datasetId": "rimes",
   "metric": "wer",
   "value": 9.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rimes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GatedHTR",
   "modelId": "",
   "dataset": "rimes",
   "datasetId": "rimes",
   "metric": "wer",
   "value": 8.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rimes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Puigcerver",
   "modelId": "",
   "dataset": "rimes",
   "datasetId": "rimes",
   "metric": "cer",
   "value": 3.21,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rimes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VAN",
   "modelId": "",
   "dataset": "rimes",
   "datasetId": "rimes",
   "metric": "cer",
   "value": 1.91,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rimes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GatedHTR",
   "modelId": "",
   "dataset": "rimes",
   "datasetId": "rimes",
   "metric": "cer",
   "value": 1.81,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//rimes",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Stable Audio Open",
   "modelId": "",
   "dataset": "audiocaps-t2a",
   "datasetId": "audiocaps-t2a",
   "metric": "fad",
   "value": 2.57,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps-t2a",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioGen Medium",
   "modelId": "",
   "dataset": "audiocaps-t2a",
   "datasetId": "audiocaps-t2a",
   "metric": "fad",
   "value": 1.82,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps-t2a",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM 2",
   "modelId": "",
   "dataset": "audiocaps-t2a",
   "datasetId": "audiocaps-t2a",
   "metric": "fad",
   "value": 1.42,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps-t2a",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM",
   "modelId": "",
   "dataset": "audiocaps",
   "datasetId": "audiocaps",
   "metric": "fad",
   "value": 4.48,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM 2-Full-Large",
   "modelId": "",
   "dataset": "audiocaps",
   "datasetId": "audiocaps",
   "metric": "fad",
   "value": 1.86,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM 2-Full",
   "modelId": "",
   "dataset": "audiocaps",
   "datasetId": "audiocaps",
   "metric": "fad",
   "value": 1.78,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TANGO",
   "modelId": "",
   "dataset": "audiocaps",
   "datasetId": "audiocaps",
   "metric": "fad",
   "value": 1.73,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM 2-AC-Large",
   "modelId": "",
   "dataset": "audiocaps",
   "datasetId": "audiocaps",
   "metric": "fad",
   "value": 1.42,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//audiocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EVA-CLIP-18B",
   "modelId": "",
   "dataset": "imagenet-zero-shot",
   "datasetId": "imagenet-zero-shot",
   "metric": "top-1",
   "value": 83.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-zero-shot",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SigLIP-SO400M",
   "modelId": "",
   "dataset": "imagenet-zero-shot",
   "datasetId": "imagenet-zero-shot",
   "metric": "top-1",
   "value": 83.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-zero-shot",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenCLIP ViT-G/14",
   "modelId": "",
   "dataset": "imagenet-zero-shot",
   "datasetId": "imagenet-zero-shot",
   "metric": "top-1",
   "value": 80.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-zero-shot",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CLIP ViT-L/14",
   "modelId": "",
   "dataset": "imagenet-zero-shot",
   "datasetId": "imagenet-zero-shot",
   "metric": "top-1",
   "value": 75.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-zero-shot",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Diffusion-QL",
   "modelId": "",
   "dataset": "d4rl-halfcheetah-medium",
   "datasetId": "d4rl-halfcheetah-medium",
   "metric": "normalized_return",
   "value": 51.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//d4rl-halfcheetah-medium",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "IQL (Implicit Q-Learning)",
   "modelId": "",
   "dataset": "d4rl-halfcheetah-medium",
   "datasetId": "d4rl-halfcheetah-medium",
   "metric": "normalized_return",
   "value": 47.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//d4rl-halfcheetah-medium",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CQL (Conservative Q-Learning)",
   "modelId": "",
   "dataset": "d4rl-halfcheetah-medium",
   "datasetId": "d4rl-halfcheetah-medium",
   "metric": "normalized_return",
   "value": 44,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//d4rl-halfcheetah-medium",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "π0 (Pi-Zero)",
   "modelId": "",
   "dataset": "libero-long",
   "datasetId": "libero-long",
   "metric": "success_rate",
   "value": 85.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//libero-long",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenVLA",
   "modelId": "",
   "dataset": "libero-long",
   "datasetId": "libero-long",
   "metric": "success_rate",
   "value": 53.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//libero-long",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Octo-Base",
   "modelId": "",
   "dataset": "libero-long",
   "datasetId": "libero-long",
   "metric": "success_rate",
   "value": 51.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//libero-long",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-Coder-32B",
   "modelId": "",
   "dataset": "mbpp-plus",
   "datasetId": "mbpp-plus",
   "metric": "pass@1",
   "value": 76.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-V3",
   "modelId": "",
   "dataset": "mbpp-plus",
   "datasetId": "mbpp-plus",
   "metric": "pass@1",
   "value": 73,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "mbpp-plus",
   "datasetId": "mbpp-plus",
   "metric": "pass@1",
   "value": 71.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-Coder-33B",
   "modelId": "",
   "dataset": "mbpp-plus",
   "datasetId": "mbpp-plus",
   "metric": "pass@1",
   "value": 66,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Marigold",
   "modelId": "",
   "dataset": "kitti-depth",
   "datasetId": "kitti-depth",
   "metric": "absrel",
   "value": 0.099,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kitti-depth",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MiDaS 3.1 (BEiT-512)",
   "modelId": "",
   "dataset": "kitti-depth",
   "datasetId": "kitti-depth",
   "metric": "absrel",
   "value": 0.058,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kitti-depth",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ZoeDepth-K",
   "modelId": "",
   "dataset": "kitti-depth",
   "datasetId": "kitti-depth",
   "metric": "absrel",
   "value": 0.053,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kitti-depth",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Depth Anything V1 (ViT-L)",
   "modelId": "",
   "dataset": "kitti-depth",
   "datasetId": "kitti-depth",
   "metric": "absrel",
   "value": 0.046,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kitti-depth",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Depth Anything V2 (ViT-L)",
   "modelId": "",
   "dataset": "kitti-depth",
   "datasetId": "kitti-depth",
   "metric": "absrel",
   "value": 0.04,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kitti-depth",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SegNet (class-level)",
   "modelId": "",
   "dataset": "dagm-2007",
   "datasetId": "dagm-2007",
   "metric": "Accuracy",
   "value": 100,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//dagm-2007",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ResNet baseline",
   "modelId": "",
   "dataset": "dagm-2007",
   "datasetId": "dagm-2007",
   "metric": "Accuracy",
   "value": 99.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//dagm-2007",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VAN",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "wer",
   "value": 16.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HTR-VT",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "wer",
   "value": 14.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HTR-ConvText",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "wer",
   "value": 12.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VAN",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "cer",
   "value": 5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HTR-VT",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "cer",
   "value": 4.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HTR-ConvText",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "cer",
   "value": 4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TrOCR-base",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "cer",
   "value": 3.42,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TrOCR-large",
   "modelId": "",
   "dataset": "iam",
   "datasetId": "iam",
   "metric": "cer",
   "value": 2.89,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//iam",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-Coder-32B",
   "modelId": "",
   "dataset": "humaneval-plus",
   "datasetId": "humaneval-plus",
   "metric": "pass@1",
   "value": 87.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-V3",
   "modelId": "",
   "dataset": "humaneval-plus",
   "datasetId": "humaneval-plus",
   "metric": "pass@1",
   "value": 86.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o",
   "modelId": "",
   "dataset": "humaneval-plus",
   "datasetId": "humaneval-plus",
   "metric": "pass@1",
   "value": 86,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-Coder-V2",
   "modelId": "",
   "dataset": "humaneval-plus",
   "datasetId": "humaneval-plus",
   "metric": "pass@1",
   "value": 82.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-Coder-33B",
   "modelId": "",
   "dataset": "humaneval-plus",
   "datasetId": "humaneval-plus",
   "metric": "pass@1",
   "value": 75,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//humaneval-plus",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HIVE-COTE 2.0",
   "modelId": "",
   "dataset": "ucr-archive",
   "datasetId": "ucr-archive",
   "metric": "mean_accuracy",
   "value": 88.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ucr-archive",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Hydra + MultiRocket",
   "modelId": "",
   "dataset": "ucr-archive",
   "datasetId": "ucr-archive",
   "metric": "mean_accuracy",
   "value": 88.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ucr-archive",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InceptionTime",
   "modelId": "",
   "dataset": "ucr-archive",
   "datasetId": "ucr-archive",
   "metric": "mean_accuracy",
   "value": 85,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ucr-archive",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternVideo2",
   "modelId": "",
   "dataset": "kinetics-400",
   "datasetId": "kinetics-400",
   "metric": "top-1",
   "value": 92.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kinetics-400",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VideoMAE V2 (ViT-g)",
   "modelId": "",
   "dataset": "kinetics-400",
   "datasetId": "kinetics-400",
   "metric": "top-1",
   "value": 90,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kinetics-400",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViViT-H",
   "modelId": "",
   "dataset": "kinetics-400",
   "datasetId": "kinetics-400",
   "metric": "top-1",
   "value": 84.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kinetics-400",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TimeSformer-L",
   "modelId": "",
   "dataset": "kinetics-400",
   "datasetId": "kinetics-400",
   "metric": "top-1",
   "value": 80.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//kinetics-400",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4",
   "modelId": "",
   "dataset": "wmt23",
   "datasetId": "wmt23",
   "metric": "comet",
   "value": 84.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wmt23",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Google Translate",
   "modelId": "",
   "dataset": "wmt23",
   "datasetId": "wmt23",
   "metric": "comet",
   "value": 83.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wmt23",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepL",
   "modelId": "",
   "dataset": "wmt23",
   "datasetId": "wmt23",
   "metric": "comet",
   "value": 83.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wmt23",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "NLLB-3.3B",
   "modelId": "",
   "dataset": "wmt23",
   "datasetId": "wmt23",
   "metric": "comet",
   "value": 81.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wmt23",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DINOv2 ViT-g/14",
   "modelId": "",
   "dataset": "imagenet-knn",
   "datasetId": "imagenet-knn",
   "metric": "top1_accuracy",
   "value": 83.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-knn",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DINOv2 ViT-L/14",
   "modelId": "",
   "dataset": "imagenet-knn",
   "datasetId": "imagenet-knn",
   "metric": "top1_accuracy",
   "value": 83.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-knn",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DINO ViT-B/16",
   "modelId": "",
   "dataset": "imagenet-knn",
   "datasetId": "imagenet-knn",
   "metric": "top1_accuracy",
   "value": 76.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-knn",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PaLI-X-55B",
   "modelId": "",
   "dataset": "ok-vqa",
   "datasetId": "ok-vqa",
   "metric": "accuracy",
   "value": 66.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ok-vqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PaLI-17B",
   "modelId": "",
   "dataset": "ok-vqa",
   "datasetId": "ok-vqa",
   "metric": "accuracy",
   "value": 64.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ok-vqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4V",
   "modelId": "",
   "dataset": "ok-vqa",
   "datasetId": "ok-vqa",
   "metric": "accuracy",
   "value": 64.28,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ok-vqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Flamingo-80B",
   "modelId": "",
   "dataset": "ok-vqa",
   "datasetId": "ok-vqa",
   "metric": "accuracy",
   "value": 57.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ok-vqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2 (FlanT5XXL)",
   "modelId": "",
   "dataset": "ok-vqa",
   "datasetId": "ok-vqa",
   "metric": "accuracy",
   "value": 44.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//ok-vqa",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EVA-02-L",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 97.15,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CoAtNet-7",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 96.38,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ConvNeXt V2-H",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 96.17,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MAE ViT-H/14",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 96.08,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SwinV2-G",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 96.01,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeiT III-H/14",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 95.94,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternImage-XL",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 95.77,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "FasterViT-6",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 95.72,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "vit-h-14",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 94.55,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AIMv2-3B",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 94.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AIMv2-1B",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 94.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViT-L/16 (IN-21K)",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 93.25,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "efficientnet-b7",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 91.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "vit-b-16",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 91.48,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "resnet-50",
   "modelId": "",
   "dataset": "cifar-100",
   "datasetId": "cifar-100",
   "metric": "accuracy",
   "value": 78.04,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "coca-finetuned",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 91,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "vit-g-14",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 90.45,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EVA-02-L",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 90.056,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EVA-Giant",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 89.79,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "InternImage-H",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 89.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SigLIP-SO400M",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 89.41,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "convnext-v2-huge",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 88.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViT-H/14 CLIP (LAION-2B)",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 88.634,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ConvNeXt-XXLarge (CLIP LAION)",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 88.622,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "vit-h-14",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 88.55,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "swin-large",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 87.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "efficientnet-v2-l",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 85.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "deit-b-distilled",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 85.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "efficientnet-b7",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 84.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "deit-b",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 83.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "convnext-v2-tiny",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 83,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "vit-l-16",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 82.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "vit-b-16",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 81.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "resnet-50-a3",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 80.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "resnet-152",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 78.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "efficientnet-b0",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 77.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "resnet-50",
   "modelId": "",
   "dataset": "imagenet-1k",
   "datasetId": "imagenet-1k",
   "metric": "top-1-accuracy",
   "value": 76.15,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//imagenet-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MusicGen-Medium",
   "modelId": "",
   "dataset": "musiccaps",
   "datasetId": "musiccaps",
   "metric": "fad",
   "value": 4.89,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//musiccaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM 2-MSD",
   "modelId": "",
   "dataset": "musiccaps",
   "datasetId": "musiccaps",
   "metric": "fad",
   "value": 4.47,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//musiccaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "MusicLM",
   "modelId": "",
   "dataset": "musiccaps",
   "datasetId": "musiccaps",
   "metric": "fad",
   "value": 4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//musiccaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM-M",
   "modelId": "",
   "dataset": "musiccaps",
   "datasetId": "musiccaps",
   "metric": "fad",
   "value": 3.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//musiccaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AudioLDM 2-Full",
   "modelId": "",
   "dataset": "musiccaps",
   "datasetId": "musiccaps",
   "metric": "fad",
   "value": 3.13,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//musiccaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SAM 2 (Hiera-L)",
   "modelId": "",
   "dataset": "sa-1b",
   "datasetId": "sa-1b",
   "metric": "miou",
   "value": 62.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//sa-1b",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SAM (ViT-H)",
   "modelId": "",
   "dataset": "sa-1b",
   "datasetId": "sa-1b",
   "metric": "miou",
   "value": 58.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//sa-1b",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "FastSAM",
   "modelId": "",
   "dataset": "sa-1b",
   "datasetId": "sa-1b",
   "metric": "miou",
   "value": 57.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//sa-1b",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EfficientSAM",
   "modelId": "",
   "dataset": "sa-1b",
   "datasetId": "sa-1b",
   "metric": "miou",
   "value": 55.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//sa-1b",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CogVLM-17B",
   "modelId": "",
   "dataset": "nocaps",
   "datasetId": "nocaps",
   "metric": "cider",
   "value": 128.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PaLI-X-55B",
   "modelId": "",
   "dataset": "nocaps",
   "datasetId": "nocaps",
   "metric": "cider",
   "value": 126.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PaLI-17B",
   "modelId": "",
   "dataset": "nocaps",
   "datasetId": "nocaps",
   "metric": "cider",
   "value": 124.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2 (FlanT5XL)",
   "modelId": "",
   "dataset": "nocaps",
   "datasetId": "nocaps",
   "metric": "cider",
   "value": 123.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BLIP-2 (OPT 2.7B)",
   "modelId": "",
   "dataset": "nocaps",
   "datasetId": "nocaps",
   "metric": "cider",
   "value": 121.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//nocaps",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BEATs",
   "modelId": "",
   "dataset": "esc-50",
   "datasetId": "esc-50",
   "metric": "accuracy",
   "value": 98.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//esc-50",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "HTS-AT",
   "modelId": "",
   "dataset": "esc-50",
   "datasetId": "esc-50",
   "metric": "accuracy",
   "value": 97,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//esc-50",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AST",
   "modelId": "",
   "dataset": "esc-50",
   "datasetId": "esc-50",
   "metric": "accuracy",
   "value": 95.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//esc-50",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "CLAP",
   "modelId": "",
   "dataset": "esc-50",
   "datasetId": "esc-50",
   "metric": "accuracy",
   "value": 93.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//esc-50",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4",
   "modelId": "",
   "dataset": "wikitablequestions",
   "datasetId": "wikitablequestions",
   "metric": "accuracy",
   "value": 75.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wikitablequestions",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Claude 3.5 Sonnet",
   "modelId": "",
   "dataset": "wikitablequestions",
   "datasetId": "wikitablequestions",
   "metric": "accuracy",
   "value": 73,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wikitablequestions",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "TAPAS-large",
   "modelId": "",
   "dataset": "wikitablequestions",
   "datasetId": "wikitablequestions",
   "metric": "accuracy",
   "value": 48.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//wikitablequestions",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViT-H/14 (JFT-300M)",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 99.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViT-L/16 (JFT-300M)",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 99.42,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "BiT-L (ResNet152x4)",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 99.37,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViT-H/14 (IN-21K)",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 99.27,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "deit-b-distilled",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 99.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViT-L/16 (IN-21K)",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 99,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EfficientNet-B8 (NoisyStudent)",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 98.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "convnext-v2-base",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 98.7,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ViT-B/16 (IN-21K)",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 98.13,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Swin-B",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 98,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "resnet-50",
   "modelId": "",
   "dataset": "cifar-10",
   "datasetId": "cifar-10",
   "metric": "accuracy",
   "value": 96.01,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//cifar-10",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SLCA (ViT-B/16)",
   "modelId": "",
   "dataset": "split-cifar100",
   "datasetId": "split-cifar100",
   "metric": "average_accuracy",
   "value": 91.53,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//split-cifar100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DualPrompt (ViT-B/16)",
   "modelId": "",
   "dataset": "split-cifar100",
   "datasetId": "split-cifar100",
   "metric": "average_accuracy",
   "value": 86.51,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//split-cifar100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "L2P (ViT-B/16)",
   "modelId": "",
   "dataset": "split-cifar100",
   "datasetId": "split-cifar100",
   "metric": "average_accuracy",
   "value": 83.86,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//split-cifar100",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Claude 3.5 Sonnet (Oct 2024)",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 91,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-Coder-32B-Instruct",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 90.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-Coder-V2-Instruct",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 89.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 89.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "gpt-4o",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 87.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4o (Aug 2024)",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 86.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen2.5-Coder-7B-Instruct",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 83.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Codestral 22B v0.1",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 78.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 4 Maverick (17B-128E)",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 77.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeepSeek-V3",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 75.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 27B IT",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 74.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 12B IT",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 73,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Llama 4 Scout (17B-16E)",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 67.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Gemma 3 4B IT",
   "modelId": "",
   "dataset": "mbpp",
   "datasetId": "mbpp",
   "metric": "pass@1",
   "value": 63.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mbpp",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4 + AlphaCodium",
   "modelId": "",
   "dataset": "codecontests",
   "datasetId": "codecontests",
   "metric": "pass@1",
   "value": 44,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//codecontests",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AlphaCode 2",
   "modelId": "",
   "dataset": "codecontests",
   "datasetId": "codecontests",
   "metric": "pass@1",
   "value": 43,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//codecontests",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "GPT-4",
   "modelId": "",
   "dataset": "codecontests",
   "datasetId": "codecontests",
   "metric": "pass@1",
   "value": 19,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//codecontests",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "P>M>F (ViT-B, DINO pretrained)",
   "modelId": "",
   "dataset": "mini-imagenet-5way5shot",
   "datasetId": "mini-imagenet-5way5shot",
   "metric": "accuracy",
   "value": 95.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mini-imagenet-5way5shot",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "FEAT (ResNet-12)",
   "modelId": "",
   "dataset": "mini-imagenet-5way5shot",
   "datasetId": "mini-imagenet-5way5shot",
   "metric": "accuracy",
   "value": 82.05,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mini-imagenet-5way5shot",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SSF (ViT-B/16)",
   "modelId": "",
   "dataset": "vtab-1k",
   "datasetId": "vtab-1k",
   "metric": "mean_accuracy",
   "value": 73.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vtab-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "VPT-Deep (ViT-B/16)",
   "modelId": "",
   "dataset": "vtab-1k",
   "datasetId": "vtab-1k",
   "metric": "mean_accuracy",
   "value": 72,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//vtab-1k",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "DeBERTa-v3-large",
   "modelId": "",
   "dataset": "glue-fill-mask",
   "datasetId": "glue-fill-mask",
   "metric": "avg-score",
   "value": 91.37,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//glue-fill-mask",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ALBERT-xxlarge-v2",
   "modelId": "",
   "dataset": "glue-fill-mask",
   "datasetId": "glue-fill-mask",
   "metric": "avg-score",
   "value": 89.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//glue-fill-mask",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "RoBERTa-large",
   "modelId": "",
   "dataset": "glue-fill-mask",
   "datasetId": "glue-fill-mask",
   "metric": "avg-score",
   "value": 88.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//glue-fill-mask",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SimpleNet",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "Image AUROC",
   "value": 99.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "simplenet",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 99.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "fastflow",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 99.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "patchcore",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 99.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "efficientad",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 99.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "PatchCore",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "Image AUROC",
   "value": 99.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "EfficientAD",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "Image AUROC",
   "value": 99.1,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "reverse-distillation",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 98.5,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "cflow-ad",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 98.3,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "draem",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 98,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "padim",
   "modelId": "",
   "dataset": "mvtec-ad",
   "datasetId": "mvtec-ad",
   "metric": "auroc",
   "value": 97.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//mvtec-ad",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "SVM + hand-crafted features",
   "modelId": "",
   "dataset": "gdxray-welds",
   "datasetId": "gdxray-welds",
   "metric": "Accuracy",
   "value": 95.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//gdxray-welds",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "ResNet50 CNN",
   "modelId": "",
   "dataset": "gdxray-welds",
   "datasetId": "gdxray-welds",
   "metric": "Accuracy",
   "value": 90.26,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//gdxray-welds",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LlamaParse Agentic",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 84.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LlamaParse Cost Effective",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 71.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Google Gemini 3 Flash",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 71,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Reducto",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 67.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Qwen 3 VL",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 62,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Azure Document Intelligence",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 59.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Extend",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 55.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Dots OCR 1.5",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 55.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Docling",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 50.6,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Google Cloud Document AI",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 50.4,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "AWS Textract",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 47.9,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "OpenAI GPT-5 Mini",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 46.8,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "LandingAI",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 45.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "Anthropic Haiku 4.5",
   "modelId": "",
   "dataset": "parsebench",
   "datasetId": "parsebench",
   "metric": "accuracy",
   "value": 45.2,
   "source": "codesota-api",
   "sourceUrl": "https://www.codesota.com/browse//parsebench",
   "accessDate": "2026-04-20",
   "verified": true,
   "verifiedBy": "CodeSOTA-API",
   "notes": "Fetched from CodeSOTA API on 2026-04-20"
  },
  {
   "model": "swin-v2-large",
   "modelId": "swin-v2-large",
   "dataset": "imagenet-v2",
   "datasetId": "imagenet-v2",
   "metric": "top-1-accuracy",
   "value": 84,
   "source": "microsoft-research",
   "sourceUrl": "https://arxiv.org/abs/2111.09883",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from microsoft-research"
  },
  {
   "model": "convnext-v2-huge",
   "modelId": "convnext-v2-huge",
   "dataset": "imagenet-v2",
   "datasetId": "imagenet-v2",
   "metric": "top-1-accuracy",
   "value": 80.5,
   "source": "meta-research",
   "sourceUrl": "https://arxiv.org/abs/2301.00808",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from meta-research"
  },
  {
   "model": "patchcore",
   "modelId": "patchcore",
   "dataset": "visa",
   "datasetId": "visa",
   "metric": "auroc",
   "value": 92.1,
   "source": "research-paper",
   "sourceUrl": "https://arxiv.org/abs/2211.14842",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from research-paper"
  },
  {
   "model": "simplenet",
   "modelId": "simplenet",
   "dataset": "visa",
   "datasetId": "visa",
   "metric": "auroc",
   "value": 95.5,
   "source": "research-paper",
   "sourceUrl": "https://arxiv.org/abs/2303.15140",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from research-paper"
  },
  {
   "model": "efficientad",
   "modelId": "efficientad",
   "dataset": "visa",
   "datasetId": "visa",
   "metric": "auroc",
   "value": 94.8,
   "source": "research-paper",
   "sourceUrl": "https://arxiv.org/abs/2303.14535",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from research-paper"
  },
  {
   "model": "o3 (high)",
   "modelId": "o3 (high)",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 98.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o4-mini (high)",
   "modelId": "o4-mini (high)",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 98.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3-mini",
   "modelId": "o3-mini",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 97.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3",
   "modelId": "o3",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 97.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o4-mini",
   "modelId": "o4-mini",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 97.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek-R1",
   "modelId": "DeepSeek-R1",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 97.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Gemini 2.5 Pro",
   "modelId": "Gemini 2.5 Pro",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 97.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1",
   "modelId": "o1",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 96.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude 3.7 Sonnet",
   "modelId": "Claude 3.7 Sonnet",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 96.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Kimi k1.5",
   "modelId": "Kimi k1.5",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 96.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek-R1-Zero",
   "modelId": "DeepSeek-R1-Zero",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 95.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek-R1-Distill-Llama-70B",
   "modelId": "DeepSeek-R1-Distill-Llama-70B",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 94.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek-R1-Distill-Qwen-32B",
   "modelId": "DeepSeek-R1-Distill-Qwen-32B",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 94.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek-V3-0324",
   "modelId": "DeepSeek-V3-0324",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 94,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "QwQ-32B",
   "modelId": "QwQ-32B",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 90.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "deepseek-v3",
   "modelId": "deepseek-v3",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 90.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-mini",
   "modelId": "o1-mini",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 90,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-4.5 Preview",
   "modelId": "GPT-4.5 Preview",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 87.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-preview",
   "modelId": "o1-preview",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 85.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-4.1",
   "modelId": "GPT-4.1",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 82.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 76.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Grok 2",
   "modelId": "Grok 2",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 76.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Llama 3.1 405B",
   "modelId": "Llama 3.1 405B",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 73.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-4 Turbo",
   "modelId": "GPT-4 Turbo",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 73.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 71.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o-mini",
   "modelId": "gpt-4o-mini",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 70.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Llama 3.1 70B",
   "modelId": "Llama 3.1 70B",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 68,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 67.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude 3 Opus",
   "modelId": "Claude 3 Opus",
   "dataset": "math",
   "datasetId": "math",
   "metric": "accuracy",
   "value": 60.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "U-Net Ensemble (Pavlov)",
   "modelId": "U-Net Ensemble (Pavlov)",
   "dataset": "severstal-steel",
   "datasetId": "severstal-steel",
   "metric": "Dice",
   "value": 0.903,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "2nd Place Solution",
   "modelId": "2nd Place Solution",
   "dataset": "severstal-steel",
   "datasetId": "severstal-steel",
   "metric": "Dice",
   "value": 0.9084,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "bestfitting (1st place ensemble)",
   "modelId": "bestfitting (1st place ensemble)",
   "dataset": "severstal-steel",
   "datasetId": "severstal-steel",
   "metric": "Dice",
   "value": 0.90883,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-preview",
   "modelId": "o1-preview",
   "dataset": "gsm8k",
   "datasetId": "gsm8k",
   "metric": "accuracy",
   "value": 97.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "gsm8k",
   "datasetId": "gsm8k",
   "metric": "accuracy",
   "value": 96.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "gsm8k",
   "datasetId": "gsm8k",
   "metric": "accuracy",
   "value": 93,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "gsm8k",
   "datasetId": "gsm8k",
   "metric": "accuracy",
   "value": 92,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "gsm8k",
   "datasetId": "gsm8k",
   "metric": "accuracy",
   "value": 91.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3",
   "modelId": "o3",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 92.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1",
   "modelId": "o1",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 91.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-45-preview",
   "modelId": "gpt-45-preview",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 90.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-preview",
   "modelId": "o1-preview",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 90.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-41",
   "modelId": "gpt-41",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 90.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o4-mini",
   "modelId": "o4-mini",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 90,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-31-405b",
   "modelId": "llama-31-405b",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 88.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "deepseek-v3",
   "modelId": "deepseek-v3",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 88.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 88.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "grok-2",
   "modelId": "grok-2",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 87.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 87.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-3-opus",
   "modelId": "claude-3-opus",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 86.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4-turbo",
   "modelId": "gpt-4-turbo",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 86.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 85.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3-mini",
   "modelId": "o3-mini",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 85.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-mini",
   "modelId": "o1-mini",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 85.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-31-70b",
   "modelId": "llama-31-70b",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 82,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o-mini",
   "modelId": "gpt-4o-mini",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 82,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "mmlu",
   "datasetId": "mmlu",
   "metric": "accuracy",
   "value": 82,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3",
   "modelId": "o3",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 82.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o4-mini",
   "modelId": "o4-mini",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 77.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1",
   "modelId": "o1",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 75.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3-mini",
   "modelId": "o3-mini",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 74.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-preview",
   "modelId": "o1-preview",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 73.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-45-preview",
   "modelId": "gpt-45-preview",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 69.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-41",
   "modelId": "gpt-41",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 66.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-mini",
   "modelId": "o1-mini",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 60,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 59.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "grok-2",
   "modelId": "grok-2",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 56,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-31-405b",
   "modelId": "llama-31-405b",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 50.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-3-opus",
   "modelId": "claude-3-opus",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 50.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 49.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4-turbo",
   "modelId": "gpt-4-turbo",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 49.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 46.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-31-70b",
   "modelId": "llama-31-70b",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 41.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o-mini",
   "modelId": "gpt-4o-mini",
   "dataset": "gpqa",
   "datasetId": "gpqa",
   "metric": "accuracy",
   "value": 40.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1-preview",
   "modelId": "o1-preview",
   "dataset": "aime-2024",
   "datasetId": "aime-2024",
   "metric": "accuracy",
   "value": 83.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-opus",
   "modelId": "claude-35-opus",
   "dataset": "aime-2024",
   "datasetId": "aime-2024",
   "metric": "accuracy",
   "value": 16,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "aime-2024",
   "datasetId": "aime-2024",
   "metric": "accuracy",
   "value": 13.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude Opus 4.7",
   "modelId": "Claude Opus 4.7",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 87.6,
   "source": "vendor",
   "sourceUrl": "https://www.anthropic.com/news/claude-opus-4-7",
   "accessDate": "2026-04-23",
   "verified": true,
   "verifiedBy": "editor",
   "notes": "Claude Code harness · Anthropic primary announcement"
  },
  {
   "model": "Claude Opus 4.5",
   "modelId": "Claude Opus 4.5",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 80.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude Opus 4.6",
   "modelId": "Claude Opus 4.6",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 80.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Gemini 3.1 Pro",
   "modelId": "Gemini 3.1 Pro",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 80.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "MiniMax M2.5",
   "modelId": "MiniMax M2.5",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 80.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-5.2 Thinking",
   "modelId": "GPT-5.2 Thinking",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 80,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude Sonnet 4.6",
   "modelId": "Claude Sonnet 4.6",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 79.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Gemini 3 Flash",
   "modelId": "Gemini 3 Flash",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 78,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude Sonnet 4.5",
   "modelId": "Claude Sonnet 4.5",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 77.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Kimi K2.5",
   "modelId": "Kimi K2.5",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 76.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-5.1",
   "modelId": "GPT-5.1",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 76.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Gemini 3 Pro",
   "modelId": "Gemini 3 Pro",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 76.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-5",
   "modelId": "GPT-5",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 74.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "MiniMax M2.1",
   "modelId": "MiniMax M2.1",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 74,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude Haiku 4.5",
   "modelId": "Claude Haiku 4.5",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 73.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude Sonnet 4",
   "modelId": "Claude Sonnet 4",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 72.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude Opus 4",
   "modelId": "Claude Opus 4",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 72.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Devstral 2",
   "modelId": "Devstral 2",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 72.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Qwen3-Coder-480B",
   "modelId": "Qwen3-Coder-480B",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 69.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "MiniMax M2",
   "modelId": "MiniMax M2",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 69.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3",
   "modelId": "o3",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 69.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o4-mini",
   "modelId": "o4-mini",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 68.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek V3.1",
   "modelId": "DeepSeek V3.1",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 66,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Kimi K2",
   "modelId": "Kimi K2",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 65.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Grok 3",
   "modelId": "Grok 3",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 63.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Gemini 2.5 Pro",
   "modelId": "Gemini 2.5 Pro",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 63.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude 3.7 Sonnet",
   "modelId": "Claude 3.7 Sonnet",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 63.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Gemini 2.5 Flash",
   "modelId": "Gemini 2.5 Flash",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 60.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek R1-0528",
   "modelId": "DeepSeek R1-0528",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 57.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o3-mini",
   "modelId": "o3-mini",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 55.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-4.1",
   "modelId": "GPT-4.1",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 54.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude 3.5 Sonnet",
   "modelId": "Claude 3.5 Sonnet",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 50.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek-R1",
   "modelId": "DeepSeek-R1",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 49.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "o1",
   "modelId": "o1",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 48.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Devstral Small 2505",
   "modelId": "Devstral Small 2505",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 46.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek V3",
   "modelId": "DeepSeek V3",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 42,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "GPT-4o",
   "modelId": "GPT-4o",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 41.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Claude 3.5 Haiku",
   "modelId": "Claude 3.5 Haiku",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 40.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DeepSeek V2.5",
   "modelId": "DeepSeek V2.5",
   "dataset": "swe-bench-verified",
   "datasetId": "swe-bench-verified",
   "metric": "resolve-rate",
   "value": 37,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "co-detr-swin-l",
   "modelId": "co-detr-swin-l",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 66,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "internimage-h",
   "modelId": "internimage-h",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 65.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Focal-Stable-DINO",
   "modelId": "Focal-Stable-DINO",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 64.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "dino-swin-l",
   "modelId": "dino-swin-l",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 63.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "EVA-02-L",
   "modelId": "EVA-02-L",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 62.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "RF-DETR-2XL",
   "modelId": "RF-DETR-2XL",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 60.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "D-FINE-X (Objects365)",
   "modelId": "D-FINE-X (Objects365)",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 59.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "yolov10-x",
   "modelId": "yolov10-x",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 57.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "RT-DETRv4-X",
   "modelId": "RT-DETRv4-X",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 57,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DINO-X Pro",
   "modelId": "DINO-X Pro",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 56,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "D-FINE-X",
   "modelId": "D-FINE-X",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 55.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "YOLOv9-E",
   "modelId": "YOLOv9-E",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 55.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "efficientdet-d7-x",
   "modelId": "efficientdet-d7-x",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 55.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "YOLO11x",
   "modelId": "YOLO11x",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 54.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "RT-DETRv3-R101",
   "modelId": "RT-DETRv3-R101",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 54.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "RT-DETRv2-X",
   "modelId": "RT-DETRv2-X",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 54.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Grounding DINO 1.5 Pro",
   "modelId": "Grounding DINO 1.5 Pro",
   "dataset": "coco",
   "datasetId": "coco",
   "metric": "mAP",
   "value": 54.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "multi-scene-f1",
   "value": 83.25,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "multilingual-f1",
   "value": 78.97,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "qwen2-vl-72b",
   "modelId": "qwen2-vl-72b",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "multi-scene-f1",
   "value": 77.95,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "internvl2-76b",
   "modelId": "internvl2-76b",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "multi-scene-f1",
   "value": 76.92,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "multi-scene-f1",
   "value": 76.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "multilingual-f1",
   "value": 73.44,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "multi-scene-f1",
   "value": 72.87,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "qwen2-vl-72b",
   "modelId": "qwen2-vl-72b",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "kie-f1",
   "value": 71.76,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "kie-f1",
   "value": 67.28,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "kie-f1",
   "value": 64.58,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "kie-f1",
   "value": 63.45,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "cc-ocr",
   "datasetId": "cc-ocr",
   "metric": "document-parsing",
   "value": 62.37,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "paddleocr",
   "modelId": "paddleocr",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.79,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "easyocr",
   "modelId": "easyocr",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.58,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "tesseract",
   "modelId": "tesseract",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.54,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "azure-ocr",
   "modelId": "azure-ocr",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.52,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o-mini",
   "modelId": "gpt-4o-mini",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.43,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.31,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "ain-7b",
   "modelId": "ain-7b",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-20-flash",
   "modelId": "gemini-20-flash",
   "dataset": "kitab-bench",
   "datasetId": "kitab-bench",
   "metric": "cer",
   "value": 0.13,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-sonnet-4",
   "modelId": "claude-sonnet-4",
   "dataset": "thaiocrbench",
   "datasetId": "thaiocrbench",
   "metric": "ted-score",
   "value": 0.84,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-25-pro",
   "modelId": "gemini-25-pro",
   "dataset": "thaiocrbench",
   "datasetId": "thaiocrbench",
   "metric": "ted-score",
   "value": 0.77,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "qwen25-vl-32b",
   "modelId": "qwen25-vl-32b",
   "dataset": "thaiocrbench",
   "datasetId": "thaiocrbench",
   "metric": "ted-score",
   "value": 0.765,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "internvl3-14b",
   "modelId": "internvl3-14b",
   "dataset": "thaiocrbench",
   "datasetId": "thaiocrbench",
   "metric": "ted-score",
   "value": 0.76,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "qwen25-vl-72b",
   "modelId": "qwen25-vl-72b",
   "dataset": "thaiocrbench",
   "datasetId": "thaiocrbench",
   "metric": "ted-score",
   "value": 0.72,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-25-pro",
   "modelId": "gemini-25-pro",
   "dataset": "mme-videoocr",
   "datasetId": "mme-videoocr",
   "metric": "total-accuracy",
   "value": 73.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "qwen25-vl-72b",
   "modelId": "qwen25-vl-72b",
   "dataset": "mme-videoocr",
   "datasetId": "mme-videoocr",
   "metric": "total-accuracy",
   "value": 69,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "internvl3-78b",
   "modelId": "internvl3-78b",
   "dataset": "mme-videoocr",
   "datasetId": "mme-videoocr",
   "metric": "total-accuracy",
   "value": 67.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "mme-videoocr",
   "datasetId": "mme-videoocr",
   "metric": "total-accuracy",
   "value": 66.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "mme-videoocr",
   "datasetId": "mme-videoocr",
   "metric": "total-accuracy",
   "value": 64.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "qwen25-vl-32b",
   "modelId": "qwen25-vl-32b",
   "dataset": "mme-videoocr",
   "datasetId": "mme-videoocr",
   "metric": "total-accuracy",
   "value": 61,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "chexpert-auc-maximizer",
   "modelId": "chexpert-auc-maximizer",
   "dataset": "chexpert",
   "datasetId": "chexpert",
   "metric": "auroc",
   "value": 93,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "biovil",
   "modelId": "biovil",
   "dataset": "chexpert",
   "datasetId": "chexpert",
   "metric": "auroc",
   "value": 89.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "chexzero",
   "modelId": "chexzero",
   "dataset": "chexpert",
   "datasetId": "chexpert",
   "metric": "auroc",
   "value": 88.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gloria",
   "modelId": "gloria",
   "dataset": "chexpert",
   "datasetId": "chexpert",
   "metric": "auroc",
   "value": 88.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "medclip",
   "modelId": "medclip",
   "dataset": "chexpert",
   "datasetId": "chexpert",
   "metric": "auroc",
   "value": 87.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "torchxrayvision",
   "modelId": "torchxrayvision",
   "dataset": "chexpert",
   "datasetId": "chexpert",
   "metric": "auroc",
   "value": 87.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "densenet-121-cxr",
   "modelId": "densenet-121-cxr",
   "dataset": "chexpert",
   "datasetId": "chexpert",
   "metric": "auroc",
   "value": 86.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "densenet-121-cxr",
   "modelId": "densenet-121-cxr",
   "dataset": "rsna-pneumonia",
   "datasetId": "rsna-pneumonia",
   "metric": "auroc",
   "value": 88.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "chexnet",
   "modelId": "chexnet",
   "dataset": "rsna-pneumonia",
   "datasetId": "rsna-pneumonia",
   "metric": "auroc",
   "value": 87.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "torchxrayvision",
   "modelId": "torchxrayvision",
   "dataset": "nih-chestxray14",
   "datasetId": "nih-chestxray14",
   "metric": "auroc",
   "value": 85.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "chexnet",
   "modelId": "chexnet",
   "dataset": "nih-chestxray14",
   "datasetId": "nih-chestxray14",
   "metric": "auroc",
   "value": 84.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "densenet-121-cxr",
   "modelId": "densenet-121-cxr",
   "dataset": "nih-chestxray14",
   "datasetId": "nih-chestxray14",
   "metric": "auroc",
   "value": 82.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "resnet-50-cxr",
   "modelId": "resnet-50-cxr",
   "dataset": "nih-chestxray14",
   "datasetId": "nih-chestxray14",
   "metric": "auroc",
   "value": 80.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "svamp",
   "datasetId": "svamp",
   "metric": "accuracy",
   "value": 93.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "svamp",
   "datasetId": "svamp",
   "metric": "accuracy",
   "value": 91.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "svamp",
   "datasetId": "svamp",
   "metric": "accuracy",
   "value": 89.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "arc-challenge",
   "datasetId": "arc-challenge",
   "metric": "accuracy",
   "value": 96.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "arc-challenge",
   "datasetId": "arc-challenge",
   "metric": "accuracy",
   "value": 96.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "arc-challenge",
   "datasetId": "arc-challenge",
   "metric": "accuracy",
   "value": 94.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "arc-challenge",
   "datasetId": "arc-challenge",
   "metric": "accuracy",
   "value": 93,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "commonsenseqa",
   "datasetId": "commonsenseqa",
   "metric": "accuracy",
   "value": 85.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "commonsenseqa",
   "datasetId": "commonsenseqa",
   "metric": "accuracy",
   "value": 83.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "commonsenseqa",
   "datasetId": "commonsenseqa",
   "metric": "accuracy",
   "value": 80.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "winogrande",
   "datasetId": "winogrande",
   "metric": "accuracy",
   "value": 87.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "winogrande",
   "datasetId": "winogrande",
   "metric": "accuracy",
   "value": 85.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "winogrande",
   "datasetId": "winogrande",
   "metric": "accuracy",
   "value": 85.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "hellaswag",
   "datasetId": "hellaswag",
   "metric": "accuracy",
   "value": 95.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gemini-15-pro",
   "modelId": "gemini-15-pro",
   "dataset": "hellaswag",
   "datasetId": "hellaswag",
   "metric": "accuracy",
   "value": 92.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "hellaswag",
   "datasetId": "hellaswag",
   "metric": "accuracy",
   "value": 89,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "hellaswag",
   "datasetId": "hellaswag",
   "metric": "accuracy",
   "value": 88,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "hotpotqa",
   "datasetId": "hotpotqa",
   "metric": "f1",
   "value": 71.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "hotpotqa",
   "datasetId": "hotpotqa",
   "metric": "f1",
   "value": 68.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "logiqa",
   "datasetId": "logiqa",
   "metric": "accuracy",
   "value": 56.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "logiqa",
   "datasetId": "logiqa",
   "metric": "accuracy",
   "value": 53.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "reclor",
   "datasetId": "reclor",
   "metric": "accuracy",
   "value": 72.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "reclor",
   "datasetId": "reclor",
   "metric": "accuracy",
   "value": 68.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "strategyqa",
   "datasetId": "strategyqa",
   "metric": "accuracy",
   "value": 82.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "strategyqa",
   "datasetId": "strategyqa",
   "metric": "accuracy",
   "value": 79.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gpt-4o",
   "modelId": "gpt-4o",
   "dataset": "mawps",
   "datasetId": "mawps",
   "metric": "accuracy",
   "value": 97.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "claude-35-sonnet",
   "modelId": "claude-35-sonnet",
   "dataset": "mawps",
   "datasetId": "mawps",
   "metric": "accuracy",
   "value": 95.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "llama-3-70b",
   "modelId": "llama-3-70b",
   "dataset": "mawps",
   "datasetId": "mawps",
   "metric": "accuracy",
   "value": 94.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "plymouth-dl-model",
   "modelId": "plymouth-dl-model",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 98,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mcbert",
   "modelId": "mcbert",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 93.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "ae-fcn",
   "modelId": "ae-fcn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 85,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "asd-swnet",
   "modelId": "asd-swnet",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "auc",
   "value": 81,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "braingт",
   "modelId": "braingт",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "auc",
   "value": 78.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "multi-atlas-dnn",
   "modelId": "multi-atlas-dnn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 78.07,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gcn",
   "modelId": "gcn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "auc",
   "value": 78,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "svm-connectivity",
   "modelId": "svm-connectivity",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "auc",
   "value": 77,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "asd-swnet",
   "modelId": "asd-swnet",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 76.52,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "maacnn",
   "modelId": "maacnn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 75.12,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "al-negat",
   "modelId": "al-negat",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 74.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "braingnn",
   "modelId": "braingnn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 73.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "gcn",
   "modelId": "gcn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 72.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "multi-task-transformer",
   "modelId": "multi-task-transformer",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 72,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "phgcl-ddgformer",
   "modelId": "phgcl-ddgformer",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 70.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "svm-connectivity",
   "modelId": "svm-connectivity",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 70.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "deep-learning-heinsfeld",
   "modelId": "deep-learning-heinsfeld",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 70,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mvs-gcn",
   "modelId": "mvs-gcn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 69.38,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mvs-gcn",
   "modelId": "mvs-gcn",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "auc",
   "value": 69.01,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "abraham-connectomes",
   "modelId": "abraham-connectomes",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 67,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "random-forest",
   "modelId": "random-forest",
   "dataset": "abide-i",
   "datasetId": "abide-i",
   "metric": "accuracy",
   "value": 63,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "deepasd",
   "modelId": "deepasd",
   "dataset": "abide-ii",
   "datasetId": "abide-ii",
   "metric": "auc",
   "value": 93,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "maacnn",
   "modelId": "maacnn",
   "dataset": "abide-ii",
   "datasetId": "abide-ii",
   "metric": "accuracy",
   "value": 72.88,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "mistral-ocr-3",
   "dataset": "internal-mistral",
   "datasetId": "internal-mistral",
   "metric": "overall-accuracy",
   "value": 94.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "mistral-ocr-3",
   "dataset": "ocr-cer-benchmark",
   "datasetId": "ocr-cer-benchmark",
   "metric": "cer",
   "value": 3.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mistral-ocr-3",
   "modelId": "mistral-ocr-3",
   "dataset": "ocr-wer-benchmark",
   "datasetId": "ocr-wer-benchmark",
   "metric": "wer",
   "value": 7.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "chexzero",
   "modelId": "chexzero",
   "dataset": "mimic-cxr",
   "datasetId": "mimic-cxr",
   "metric": "auroc",
   "value": 89.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "torchxrayvision",
   "modelId": "torchxrayvision",
   "dataset": "mimic-cxr",
   "datasetId": "mimic-cxr",
   "metric": "auroc",
   "value": 86.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "convirt",
   "modelId": "convirt",
   "dataset": "mimic-cxr",
   "datasetId": "mimic-cxr",
   "metric": "auroc",
   "value": 85.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "rad-dino",
   "modelId": "rad-dino",
   "dataset": "vindr-cxr",
   "datasetId": "vindr-cxr",
   "metric": "auroc",
   "value": 91.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "torchxrayvision",
   "modelId": "torchxrayvision",
   "dataset": "vindr-cxr",
   "datasetId": "vindr-cxr",
   "metric": "auroc",
   "value": 87.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "torchxrayvision",
   "modelId": "torchxrayvision",
   "dataset": "padchest",
   "datasetId": "padchest",
   "metric": "auroc",
   "value": 84.6,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "densenet-121-cxr",
   "modelId": "densenet-121-cxr",
   "dataset": "covid-chestxray",
   "datasetId": "covid-chestxray",
   "metric": "auroc",
   "value": 94.7,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "torchxrayvision",
   "modelId": "torchxrayvision",
   "dataset": "covid-chestxray",
   "datasetId": "covid-chestxray",
   "metric": "auroc",
   "value": 93.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "yolov8-weld",
   "modelId": "yolov8-weld",
   "dataset": "weld-defect-xray",
   "datasetId": "weld-defect-xray",
   "metric": "map",
   "value": 87.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "defectdet-resnet",
   "modelId": "defectdet-resnet",
   "dataset": "neu-det",
   "datasetId": "neu-det",
   "metric": "map",
   "value": 78.4,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mistral-ocr-2512",
   "modelId": "mistral-ocr-2512",
   "dataset": "codesota-verification",
   "datasetId": "codesota-verification",
   "metric": "pages-per-second",
   "value": 1.22,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "ONE-PEACE",
   "modelId": "ONE-PEACE",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 63,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "internimage-h",
   "modelId": "internimage-h",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 62.9,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "ViT-Adapter-L (BEiT-3)",
   "modelId": "ViT-Adapter-L (BEiT-3)",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 62.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "ViT-CoMer-L",
   "modelId": "ViT-CoMer-L",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 62.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "DINOv2 ViT-g/14 + Mask2Former",
   "modelId": "DINOv2 ViT-g/14 + Mask2Former",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 60.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "EVA-02-L + UperNet",
   "modelId": "EVA-02-L + UperNet",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 60.1,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "EoMT-L (DINOv2)",
   "modelId": "EoMT-L (DINOv2)",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 59.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "OneFormer (DiNAT-L)",
   "modelId": "OneFormer (DiNAT-L)",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 58.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "mask2former-swin-l",
   "modelId": "mask2former-swin-l",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 57.3,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Swin-L + UperNet",
   "modelId": "Swin-L + UperNet",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 53.5,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "SegMAN-L",
   "modelId": "SegMAN-L",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 53.2,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "SegFormer-B5",
   "modelId": "SegFormer-B5",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 51.8,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "SeMask-L",
   "modelId": "SeMask-L",
   "dataset": "ade20k",
   "datasetId": "ade20k",
   "metric": "mIoU",
   "value": 49.35,
   "source": "src",
   "sourceUrl": "",
   "accessDate": "2026-04-20",
   "verified": false,
   "verifiedBy": "auto",
   "notes": "Non-API entry from src"
  },
  {
   "model": "Codex / GPT-5.5",
   "modelId": "codex-gpt-5-5",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 82,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 1 on terminal-bench@2.0. Agent: Codex. Model: GPT-5.5. Date: 2026-04-23. Official leaderboard reports 82.0% +/- 2.2."
  },
  {
   "model": "ForgeCode / GPT-5.4",
   "modelId": "forgecode-gpt-5-4",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 81.8,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 2 on terminal-bench@2.0. Agent org: ForgeCode. Model org: OpenAI. Date: 2026-03-12. Official leaderboard reports 81.8% +/- 2.0."
  },
  {
   "model": "TongAgents / Gemini 3.1 Pro",
   "modelId": "tongagents-gemini-3-1-pro",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 80.2,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 3 on terminal-bench@2.0. Agent org: BIGAI. Model org: Google. Date: 2026-03-13. Official leaderboard reports 80.2% +/- 2.6."
  },
  {
   "model": "ForgeCode / Claude Opus 4.6",
   "modelId": "forgecode-claude-opus-4-6",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 79.8,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 4 on terminal-bench@2.0. Agent org: ForgeCode. Model org: Anthropic. Date: 2026-03-12. Official leaderboard reports 79.8% +/- 1.6."
  },
  {
   "model": "SageAgent / GPT-5.3-Codex",
   "modelId": "sageagent-gpt-5-3-codex",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 78.4,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 5 on terminal-bench@2.0. Agent org: OpenSage. Model org: OpenAI. Date: 2026-03-13. Official leaderboard reports 78.4% +/- 2.2."
  },
  {
   "model": "ForgeCode / Gemini 3.1 Pro",
   "modelId": "forgecode-gemini-3-1-pro",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 78.4,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 6 on terminal-bench@2.0. Agent org: ForgeCode. Model org: Google. Date: 2026-03-02. Official leaderboard reports 78.4% +/- 1.8."
  },
  {
   "model": "Droid / GPT-5.3-Codex",
   "modelId": "droid-gpt-5-3-codex",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 77.3,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 7 on terminal-bench@2.0. Agent org: Factory. Model org: OpenAI. Date: 2026-02-24. Official leaderboard reports 77.3% +/- 2.2."
  },
  {
   "model": "Capy / Claude Opus 4.6",
   "modelId": "capy-claude-opus-4-6",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 75.3,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 8 on terminal-bench@2.0. Agent org: Capy. Model org: Anthropic. Date: 2026-03-12. Official leaderboard reports 75.3% +/- 2.4."
  },
  {
   "model": "Simple Codex / GPT-5.3-Codex",
   "modelId": "simple-codex-gpt-5-3-codex",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 75.1,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 9 on terminal-bench@2.0. Agent org: OpenAI. Model org: OpenAI. Date: 2026-02-06. Official leaderboard reports 75.1% +/- 2.4."
  },
  {
   "model": "Terminus-KIRA / Gemini 3.1 Pro",
   "modelId": "terminus-kira-gemini-3-1-pro",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 74.8,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 10 on terminal-bench@2.0. Agent org: KRAFTON AI. Model org: Google. Date: 2026-02-23. Official leaderboard reports 74.8% +/- 2.6."
  },
  {
   "model": "Terminus-KIRA / Claude Opus 4.6",
   "modelId": "terminus-kira-claude-opus-4-6",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 74.7,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 11 on terminal-bench@2.0. Agent org: KRAFTON AI. Model org: Anthropic. Date: 2026-02-22. Official leaderboard reports 74.7% +/- 2.6."
  },
  {
   "model": "Mux / GPT-5.3-Codex",
   "modelId": "mux-gpt-5-3-codex",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 74.6,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 12 on terminal-bench@2.0. Agent org: Coder. Model org: OpenAI. Date: 2026-03-06. Official leaderboard reports 74.6% +/- 2.5."
  },
  {
   "model": "MAYA-V2 / Claude 4.6 Opus",
   "modelId": "maya-v2-claude-4-6-opus",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 72.1,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 13 on terminal-bench@2.0. Agent org: ADYA. Model org: Anthropic. Date: 2026-03-12. Official leaderboard reports 72.1% +/- 2.2."
  },
  {
   "model": "TongAgents / Claude Opus 4.6",
   "modelId": "tongagents-claude-opus-4-6",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 71.9,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 14 on terminal-bench@2.0. Agent org: Bigai. Model org: Anthropic. Date: 2026-02-22. Official leaderboard reports 71.9% +/- 2.7."
  },
  {
   "model": "Junie CLI / Multiple",
   "modelId": "junie-cli-multiple",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 71,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 15 on terminal-bench@2.0. Agent org: JetBrains. Model org: Multiple. Date: 2026-03-07. Official leaderboard reports 71.0% +/- 2.9."
  },
  {
   "model": "CodeBrain-1 / GPT-5.3-Codex",
   "modelId": "codebrain-1-gpt-5-3-codex",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 70.3,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 16 on terminal-bench@2.0. Agent org: Feeling AI. Model org: OpenAI. Date: 2026-02-10. Official leaderboard reports 70.3% +/- 2.6."
  },
  {
   "model": "Droid / Claude Opus 4.6",
   "modelId": "droid-claude-opus-4-6",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 69.9,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 17 on terminal-bench@2.0. Agent org: Factory. Model org: Anthropic. Date: 2026-02-05. Official leaderboard reports 69.9% +/- 2.5."
  },
  {
   "model": "Ante / Gemini 3 Pro",
   "modelId": "ante-gemini-3-pro",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 69.4,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 18 on terminal-bench@2.0. Agent org: Antigma Labs. Model org: Google. Date: 2026-01-06. Official leaderboard reports 69.4% +/- 2.1."
  },
  {
   "model": "IndusAGI Coding Agent / GPT-5.3-Codex",
   "modelId": "indusagi-coding-agent-gpt-5-3-codex",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 69.1,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 19 on terminal-bench@2.0. Agent org: Varun Israni (SoloVpx). Model org: OpenAI. Date: 2026-03-18. Official leaderboard reports 69.1% +/- 2.3."
  },
  {
   "model": "Crux / Claude Opus 4.6",
   "modelId": "crux-claude-opus-4-6",
   "dataset": "terminal-bench-2",
   "datasetId": "terminal-bench-2",
   "metric": "accuracy",
   "value": 66.9,
   "source": "terminal-bench-official",
   "sourceUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
   "accessDate": "2026-04-27",
   "verified": true,
   "verifiedBy": "Terminal-Bench team",
   "notes": "Rank 20 on terminal-bench@2.0. Agent org: Roam. Model org: Anthropic. Date: 2026-02-23. Official leaderboard reports 66.9% +/- N/A."
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5637-text-edit-distance",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "PP-StructureV3",
   "model": "PP-StructureV3",
   "modelId": "PP-StructureV3",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.145,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2507.05595",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.0 Edit Distance; lower is better.; Table 1 of the PaddleOCR 3.0 Technical Report reports OmniDocBench document parsing edit distance for PP-StructureV3: EN 0.145 and ZH 0.206. Imported the English split as the leaderboard score following existing OmniDocBench v1.0 convention; Chinese split is recorded here but not imported because there is no separate local split leaderboard. Metric: Edit Distance (lower is better).; PWC evaluation id 5637; paper: PaddleOCR 3.0 Technical Report",
   "checkpoint_or_api_version": "PP-StructureV3",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-07-08",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Edit Distance": "0.145"
   },
   "artifacts": {
    "pwc_dataset_id": "139",
    "pwc_evaluation_id": "5637",
    "paper_arxiv_id": "2507.05595",
    "paper_id": "51740",
    "code_url": "https://github.com/PaddlePaddle/PaddleOCR",
    "external_source_url": null,
    "best_rank": 4,
    "original_metric": "Edit Distance",
    "original_value": "0.145"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5612-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "sensenova-u1-a3b-mot",
   "model": "sensenova-u1-a3b-mot",
   "modelId": "sensenova-u1-a3b-mot",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 919,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2605.12500",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC OCRBench Score normalized to the 0-1000 convention.; Paper Table 3; SenseNova-U1-A3B-MoT Think mode on OCRBench.; PWC evaluation id 5612; paper: SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture",
   "checkpoint_or_api_version": "SenseNova-U1-A3B-MoT",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-05-12",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "91.90"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5612",
    "paper_arxiv_id": "2605.12500",
    "paper_id": "83227",
    "code_url": "https://github.com/OpenSenseNova/SenseNova-U1",
    "external_source_url": null,
    "best_rank": 26,
    "original_metric": "Score",
    "original_value": "91.90"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5587-overall-zh-private",
   "task_id": "ocr",
   "benchmark_id": "ocrbench-v2",
   "model_id": "ovis2-5-9b",
   "model": "ovis2-5-9b",
   "modelId": "ovis2-5-9b",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 58,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2508.11737",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OCRBench v2 Chinese Score.; Table 6, OCR & chart; OCRBench v2 Chinese split. Source/provenance: Ovis2.5 Technical Report; source arXiv paper https://arxiv.org/abs/2508.11737; official HF model URL https://huggingface.co/AIDC-AI/Ovis2.5-9B.; PWC evaluation id 5587; paper: Ovis2.5 Technical Report",
   "checkpoint_or_api_version": "Ovis2.5-9B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-08-15",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Chinese Score": "58.0"
   },
   "artifacts": {
    "pwc_dataset_id": "156",
    "pwc_evaluation_id": "5587",
    "paper_arxiv_id": "2508.11737",
    "paper_id": "52395",
    "code_url": "https://github.com/AIDC-AI/Ovis",
    "external_source_url": "https://arxiv.org/abs/2508.11737",
    "best_rank": 1,
    "original_metric": "Chinese Score",
    "original_value": "58.0"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5586-overall-en-private",
   "task_id": "ocr",
   "benchmark_id": "ocrbench-v2",
   "model_id": "ovis2-5-9b",
   "model": "ovis2-5-9b",
   "modelId": "ovis2-5-9b",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 63.4,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2508.11737",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OCRBench v2 English Score.; Table 6, OCR & chart; OCRBench v2 English split. Source/provenance: Ovis2.5 Technical Report; source arXiv paper https://arxiv.org/abs/2508.11737; official HF model URL https://huggingface.co/AIDC-AI/Ovis2.5-9B.; PWC evaluation id 5586; paper: Ovis2.5 Technical Report",
   "checkpoint_or_api_version": "Ovis2.5-9B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-08-15",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "English Score": "63.4"
   },
   "artifacts": {
    "pwc_dataset_id": "156",
    "pwc_evaluation_id": "5586",
    "paper_arxiv_id": "2508.11737",
    "paper_id": "52395",
    "code_url": "https://github.com/AIDC-AI/Ovis",
    "external_source_url": "https://arxiv.org/abs/2508.11737",
    "best_rank": 1,
    "original_metric": "English Score",
    "original_value": "63.4"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5583-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "ovis2-5-9b",
   "model": "ovis2-5-9b",
   "modelId": "ovis2-5-9b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 879,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2508.11737",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC OCRBench Score normalized to the 0-1000 convention.; Table 3, OpenCompass suite; OCRBench (OCR). Source/provenance: Ovis2.5 Technical Report; source arXiv paper https://arxiv.org/abs/2508.11737; official HF model URL https://huggingface.co/AIDC-AI/Ovis2.5-9B.; PWC evaluation id 5583; paper: Ovis2.5 Technical Report",
   "checkpoint_or_api_version": "Ovis2.5-9B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-08-15",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "87.9"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5583",
    "paper_arxiv_id": "2508.11737",
    "paper_id": "52395",
    "code_url": "https://github.com/AIDC-AI/Ovis",
    "external_source_url": "https://arxiv.org/abs/2508.11737",
    "best_rank": 30,
    "original_metric": "Score",
    "original_value": "87.9"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5563-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen3-6-27b",
   "model": "qwen3-6-27b",
   "modelId": "qwen3-6-27b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 894,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/Qwen/Qwen3.6-27B",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC OCRBench Score normalized to the 0-1000 convention.; OCRBench row from Qwen3.6 model-card Vision benchmark table; imported as configured OCRBench Score. Source: Qwen3.6-27B Hugging Face model card benchmark table (https://huggingface.co/Qwen/Qwen3.6-27B).; PWC evaluation id 5563; paper: Qwen3.6",
   "checkpoint_or_api_version": "Qwen3.6-27B",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-04-21",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "89.4"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5563",
    "paper_arxiv_id": null,
    "paper_id": "83277",
    "code_url": "https://github.com/QwenLM/Qwen3.6",
    "external_source_url": "https://huggingface.co/Qwen/Qwen3.6-27B",
    "best_rank": 29,
    "original_metric": "Score",
    "original_value": "89.4"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5562-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen3-6-35b-a3b",
   "model": "qwen3-6-35b-a3b",
   "modelId": "qwen3-6-35b-a3b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 900,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/Qwen/Qwen3.6-27B",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC OCRBench Score normalized to the 0-1000 convention.; OCRBench row from Qwen3.6 model-card Vision benchmark table; imported as configured OCRBench Score. Source: Qwen3.6-27B Hugging Face model card benchmark table (https://huggingface.co/Qwen/Qwen3.6-27B).; PWC evaluation id 5562; paper: Qwen3.6",
   "checkpoint_or_api_version": "Qwen3.6-35B-A3B",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-04-21",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "90.0"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5562",
    "paper_arxiv_id": null,
    "paper_id": "83277",
    "code_url": "https://github.com/QwenLM/Qwen3.6",
    "external_source_url": "https://huggingface.co/Qwen/Qwen3.6-27B",
    "best_rank": 28,
    "original_metric": "Score",
    "original_value": "90.0"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5378-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen2-5-vl-3b",
   "model": "qwen2-5-vl-3b",
   "modelId": "qwen2-5-vl-3b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 797,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2502.13923",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "paper table; source label OCRBench; metric reported as Score. Imported while expanding from ScreenSpot-Pro source papers.; PWC evaluation id 5378; paper: Qwen2.5-VL Technical Report",
   "checkpoint_or_api_version": "Qwen2.5-VL-3B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-02-19",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "797"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5378",
    "paper_arxiv_id": "2502.13923",
    "paper_id": "44599",
    "code_url": "https://github.com/qwenlm/qwen2.5-vl",
    "external_source_url": null,
    "best_rank": 23,
    "original_metric": "Score",
    "original_value": "797"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5377-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "Qwen2.5-VL-7B",
   "model": "Qwen2.5-VL-7B",
   "modelId": "Qwen2.5-VL-7B",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 864,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2502.13923",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "paper table; source label OCRBench; metric reported as Score. Imported while expanding from ScreenSpot-Pro source papers.; PWC evaluation id 5377; paper: Qwen2.5-VL Technical Report",
   "checkpoint_or_api_version": "Qwen2.5-VL-7B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-02-19",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "864"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5377",
    "paper_arxiv_id": "2502.13923",
    "paper_id": "44599",
    "code_url": "https://github.com/qwenlm/qwen2.5-vl",
    "external_source_url": null,
    "best_rank": 15,
    "original_metric": "Score",
    "original_value": "864"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5329-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen3-5-omni-plus",
   "model": "qwen3-5-omni-plus",
   "modelId": "qwen3-5-omni-plus",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 913,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2604.15804",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC OCRBench Score normalized to the 0-1000 convention.; Paper Table 6, Vision->Text, OCRBench document understanding benchmark.; PWC evaluation id 5329; paper: Qwen3.5-Omni Technical Report",
   "checkpoint_or_api_version": "Qwen3.5-Omni-Plus",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-04-17",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "91.3"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5329",
    "paper_arxiv_id": "2604.15804",
    "paper_id": "57552",
    "code_url": null,
    "external_source_url": null,
    "best_rank": 27,
    "original_metric": "Score",
    "original_value": "91.3"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5183-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "minicpm-llama3-v-2-5",
   "model": "minicpm-llama3-v-2-5",
   "modelId": "minicpm-llama3-v-2-5",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 725,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2408.01800",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Paper Table 5 OCR benchmark result for MiniCPM-Llama3-V 2.5; source reports OCRBench score.; PWC evaluation id 5183; paper: MiniCPM-V: A GPT-4V Level MLLM on Your Phone",
   "checkpoint_or_api_version": "MiniCPM-Llama3-V 2.5",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2024-08-03",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "725"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5183",
    "paper_arxiv_id": "2408.01800",
    "paper_id": "35515",
    "code_url": "https://github.com/OpenBMB/MiniCPM-o",
    "external_source_url": null,
    "best_rank": 25,
    "original_metric": "Score",
    "original_value": "725"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5083-overall-zh-private",
   "task_id": "ocr",
   "benchmark_id": "ocrbench-v2",
   "model_id": "intern-s1-pro",
   "model": "intern-s1-pro",
   "modelId": "intern-s1-pro",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 60.6,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/internlm/Intern-S1-Pro",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OCRBench v2 Chinese Score.; Reported in the Intern-S1-Pro paper and Hugging Face model card performance table as OCRBench V2 (ENG / CHN). OCRBench V2 is evaluated with the non-thinking configuration; scores are English 60.1 and Chinese 60.6.; PWC evaluation id 5083; paper: Intern-S1-Pro: Scientific Multimodal Foundation Model at Trillion Scale",
   "checkpoint_or_api_version": "Intern-S1-Pro",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-03-26",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Chinese Score": "60.6",
    "English Score": "60.1"
   },
   "artifacts": {
    "pwc_dataset_id": "156",
    "pwc_evaluation_id": "5083",
    "paper_arxiv_id": "2603.25040",
    "paper_id": "57021",
    "code_url": null,
    "external_source_url": "https://huggingface.co/internlm/Intern-S1-Pro",
    "best_rank": 2,
    "original_metric": "Chinese Score",
    "original_value": "60.6"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5083-overall-en-private",
   "task_id": "ocr",
   "benchmark_id": "ocrbench-v2",
   "model_id": "intern-s1-pro",
   "model": "intern-s1-pro",
   "modelId": "intern-s1-pro",
   "dataset": "ocrbench-v2",
   "datasetId": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 60.1,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/internlm/Intern-S1-Pro",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OCRBench v2 English Score.; Reported in the Intern-S1-Pro paper and Hugging Face model card performance table as OCRBench V2 (ENG / CHN). OCRBench V2 is evaluated with the non-thinking configuration; scores are English 60.1 and Chinese 60.6.; PWC evaluation id 5083; paper: Intern-S1-Pro: Scientific Multimodal Foundation Model at Trillion Scale",
   "checkpoint_or_api_version": "Intern-S1-Pro",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-03-26",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Chinese Score": "60.6",
    "English Score": "60.1"
   },
   "artifacts": {
    "pwc_dataset_id": "156",
    "pwc_evaluation_id": "5083",
    "paper_arxiv_id": "2603.25040",
    "paper_id": "57021",
    "code_url": null,
    "external_source_url": "https://huggingface.co/internlm/Intern-S1-Pro",
    "best_rank": 2,
    "original_metric": "English Score",
    "original_value": "60.1"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5023-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "Qwen2.5-VL 72B",
   "model": "Qwen2.5-VL 72B",
   "modelId": "Qwen2.5-VL 72B",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 885,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2502.13923",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Table 5, OCRBench. Source: Qwen2.5-VL Technical Report (arXiv:2502.13923). Model: Qwen2.5-VL-72B.; PWC evaluation id 5023; paper: Qwen2.5-VL Technical Report",
   "checkpoint_or_api_version": "Qwen2.5-VL-72B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-02-19",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "885"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "5023",
    "paper_arxiv_id": "2502.13923",
    "paper_id": "44599",
    "code_url": "https://github.com/qwenlm/qwen2.5-vl",
    "external_source_url": null,
    "best_rank": 6,
    "original_metric": "Score",
    "original_value": "885"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-5017-text-edit-distance",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "Qwen2.5-VL 72B",
   "model": "Qwen2.5-VL 72B",
   "modelId": "Qwen2.5-VL 72B",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.226,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2502.13923",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.0 Edit Distance; lower is better.; Table 5, OmniDocBench English edit distance from the reported en/zh pair 0.226/0.324; Chinese split not imported because there is no separate local leaderboard. Source: Qwen2.5-VL Technical Report (arXiv:2502.13923). Model: Qwen2.5-VL-72B.; PWC evaluation id 5017; paper: Qwen2.5-VL Technical Report",
   "checkpoint_or_api_version": "Qwen2.5-VL-72B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-02-19",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Edit Distance": "0.226"
   },
   "artifacts": {
    "pwc_dataset_id": "139",
    "pwc_evaluation_id": "5017",
    "paper_arxiv_id": "2502.13923",
    "paper_id": "44599",
    "code_url": "https://github.com/qwenlm/qwen2.5-vl",
    "external_source_url": null,
    "best_rank": 5,
    "original_metric": "Edit Distance",
    "original_value": "0.226"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4979-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "olmocr",
   "model": "olmocr",
   "modelId": "olmocr",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 75.5,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2502.18443",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Paper Table 4, olmOCR-Bench overall unit-test pass rate for Ours (v0.1.75 Anchored). Category scores reported in the paper: AR 74.9, OSM 71.2, TA 71.0, OS 42.2, HF 94.5, MC 78.3, LTT 73.3, Base 98.3.; PWC evaluation id 4979; paper: olmOCR: Unlocking Trillions of Tokens in PDFs with Vision Language Models",
   "checkpoint_or_api_version": "olmOCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-02-25",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "75.5"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4979",
    "paper_arxiv_id": "2502.18443",
    "paper_id": "45019",
    "code_url": "https://github.com/allenai/olmocr",
    "external_source_url": null,
    "best_rank": 16,
    "original_metric": "Accuracy",
    "original_value": "75.5"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4968-text-edit-distance",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "lightonocr-1b-1025",
   "model": "lightonocr-1b-1025",
   "modelId": "lightonocr-1b-1025",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.234,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2601.14251",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.0 Edit Distance; lower is better.; OmniDocBench v1.0 EN Overall Edit 0.234 from LightOnOCR arXiv Appendix C.2 / Table 6; metric: Edit Distance (lower is better).; PWC evaluation id 4968; paper: LightOnOCR: A 1B End-to-End Multilingual Vision-Language Model for State-of-the-Art OCR",
   "checkpoint_or_api_version": "LightOnOCR-1B-1025",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-01-20",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Edit Distance": "0.234"
   },
   "artifacts": {
    "pwc_dataset_id": "139",
    "pwc_evaluation_id": "4968",
    "paper_arxiv_id": "2601.14251",
    "paper_id": "55399",
    "code_url": null,
    "external_source_url": null,
    "best_rank": 6,
    "original_metric": "Edit Distance",
    "original_value": "0.234"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4967-text-edit-distance",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "dots-ocr",
   "model": "dots-ocr",
   "modelId": "dots-ocr",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.125,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2512.02498",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.0 Edit Distance; lower is better.; OmniDocBench v1.0 EN Overall Edit 0.125 from the dots.ocr paper, Hugging Face card, and GitHub blog end-to-end table; metric: Edit Distance (lower is better).; PWC evaluation id 4967; paper: dots.ocr: Multilingual Document Layout Parsing in a Single Vision-Language Model",
   "checkpoint_or_api_version": "dots.ocr",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-12-02",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Edit Distance": "0.125"
   },
   "artifacts": {
    "pwc_dataset_id": "139",
    "pwc_evaluation_id": "4967",
    "paper_arxiv_id": "2512.02498",
    "paper_id": "64224",
    "code_url": "https://github.com/rednote-hilab/dots.ocr",
    "external_source_url": null,
    "best_rank": 3,
    "original_metric": "Edit Distance",
    "original_value": "0.125"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4966-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "infinity-parser2-pro",
   "model": "infinity-parser2-pro",
   "modelId": "infinity-parser2-pro",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 862,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/infly/Infinity-Parser2-Pro",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "OCRBench full benchmark score from the Infinity-Parser2-Pro Hugging Face card / GitHub performance table. Source reports 86.20 on a 0-100 scale; stored as 862.0 on the 0-1000 OCRBench Score convention used by existing rows.; PWC evaluation id 4966; paper: Infinity-Parser2-Pro",
   "checkpoint_or_api_version": "Infinity-Parser2-Pro",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-05-11",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "862.0"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "4966",
    "paper_arxiv_id": null,
    "paper_id": "83158",
    "code_url": null,
    "external_source_url": "https://huggingface.co/infly/Infinity-Parser2-Pro",
    "best_rank": 16,
    "original_metric": "Score",
    "original_value": "862.0"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4957-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "deepseek-ocr-2",
   "model": "deepseek-ocr-2",
   "modelId": "deepseek-ocr-2",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 91.09,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2601.20552",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; OmniDocBench v1.5 Overall 91.09 from DeepSeek-OCR 2 paper Table 1 main results, V-token max=1120; metric: Accuracy.; PWC evaluation id 4957; paper: DeepSeek-OCR 2: Visual Causal Flow",
   "checkpoint_or_api_version": "DeepSeek-OCR-2",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-01-28",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "91.09"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "4957",
    "paper_arxiv_id": "2601.20552",
    "paper_id": "55551",
    "code_url": "https://github.com/deepseek-ai/DeepSeek-OCR-2",
    "external_source_url": null,
    "best_rank": 5,
    "original_metric": "Accuracy",
    "original_value": "91.09"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4956-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "GLM-OCR",
   "model": "GLM-OCR",
   "modelId": "GLM-OCR",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 94.62,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.10910",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; Overall 94.62 on OmniDocBench v1.5 from GLM-OCR arXiv Table 4 (Table 3 rounds to 94.6); metric: Accuracy.; PWC evaluation id 4956; paper: GLM-OCR Technical Report",
   "checkpoint_or_api_version": "GLM-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-11",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "94.62"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "4956",
    "paper_arxiv_id": "2603.10910",
    "paper_id": "59560",
    "code_url": null,
    "external_source_url": null,
    "best_rank": 1,
    "original_metric": "Accuracy",
    "original_value": "94.62"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4955-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "firered-ocr-2b",
   "model": "firered-ocr-2b",
   "modelId": "firered-ocr-2b",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 92.94,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.01840",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; Overall 92.94 on OmniDocBench v1.5 from the FireRed-OCR Hugging Face card, GitHub README, and arXiv paper Tables 1-2; metric: Accuracy.; PWC evaluation id 4955; paper: FireRed-OCR Technical Report",
   "checkpoint_or_api_version": "FireRed-OCR-2B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-02",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "92.94"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "4955",
    "paper_arxiv_id": "2603.01840",
    "paper_id": "56414",
    "code_url": "https://github.com/FireRedTeam/FireRed-OCR",
    "external_source_url": null,
    "best_rank": 3,
    "original_metric": "Accuracy",
    "original_value": "92.94"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4952-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "firered-ocr",
   "model": "firered-ocr",
   "modelId": "firered-ocr",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 70.2,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.01840",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for FireRedTeam/FireRed-OCR. Source note: Headers & Footers category excluded; metric: Accuracy.; PWC evaluation id 4952; paper: FireRed-OCR Technical Report",
   "checkpoint_or_api_version": "FireRed-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-02",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "70.2"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4952",
    "paper_arxiv_id": "2603.01840",
    "paper_id": "56414",
    "code_url": "https://github.com/FireRedTeam/FireRed-OCR",
    "external_source_url": null,
    "best_rank": 18,
    "original_metric": "Accuracy",
    "original_value": "70.2"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4951-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "GLM-OCR",
   "model": "GLM-OCR",
   "modelId": "GLM-OCR",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 75.2,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.10910",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for zai-org/GLM-OCR. Source note: Headers & Footers category excluded; evaluated via ZAI API; metric: Accuracy.; PWC evaluation id 4951; paper: GLM-OCR Technical Report",
   "checkpoint_or_api_version": "GLM-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-11",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "75.2"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4951",
    "paper_arxiv_id": "2603.10910",
    "paper_id": "59560",
    "code_url": null,
    "external_source_url": null,
    "best_rank": 17,
    "original_metric": "Accuracy",
    "original_value": "75.2"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4950-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "DeepSeek-OCR",
   "model": "DeepSeek-OCR",
   "modelId": "DeepSeek-OCR",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 75.7,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2510.18234",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for deepseek-ai/DeepSeek-OCR. Source: olmOCR-Bench GitHub leaderboard entry and DeepSeek-OCR arXiv-linked model card; metric: Accuracy.; PWC evaluation id 4950; paper: DeepSeek-OCR: Contexts Optical Compression",
   "checkpoint_or_api_version": "DeepSeek-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-10-21",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "75.7"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4950",
    "paper_arxiv_id": "2510.18234",
    "paper_id": "53719",
    "code_url": "https://github.com/deepseek-ai/DeepSeek-OCR",
    "external_source_url": null,
    "best_rank": 15,
    "original_metric": "Accuracy",
    "original_value": "75.7"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4949-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "lightonocr-1b-1025",
   "model": "lightonocr-1b-1025",
   "modelId": "lightonocr-1b-1025",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 76.1,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2601.14251",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for lightonai/LightOnOCR-1B-1025. Source note: Headers & Footers category excluded; metric: Accuracy.; PWC evaluation id 4949; paper: LightOnOCR: A 1B End-to-End Multilingual Vision-Language Model for State-of-the-Art OCR",
   "checkpoint_or_api_version": "LightOnOCR-1B-1025",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-01-20",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "76.1"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4949",
    "paper_arxiv_id": "2601.14251",
    "paper_id": "55399",
    "code_url": null,
    "external_source_url": null,
    "best_rank": 14,
    "original_metric": "Accuracy",
    "original_value": "76.1"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4948-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "deepseek-ocr-2",
   "model": "deepseek-ocr-2",
   "modelId": "deepseek-ocr-2",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 76.3,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2601.20552",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for deepseek-ai/DeepSeek-OCR-2. Source: HF leaderboard entry and DeepSeek-OCR-2 arXiv-linked model card; metric: Accuracy.; PWC evaluation id 4948; paper: DeepSeek-OCR 2: Visual Causal Flow",
   "checkpoint_or_api_version": "DeepSeek-OCR-2",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-01-28",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "76.3"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4948",
    "paper_arxiv_id": "2601.20552",
    "paper_id": "55551",
    "code_url": "https://github.com/deepseek-ai/DeepSeek-OCR-2",
    "external_source_url": null,
    "best_rank": 13,
    "original_metric": "Accuracy",
    "original_value": "76.3"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4947-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "dots-ocr",
   "model": "dots-ocr",
   "modelId": "dots-ocr",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 79.1,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2512.02498",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for rednote-hilab/dots.ocr. Source: dots.ocr technical report / HF leaderboard entry; metric: Accuracy.; PWC evaluation id 4947; paper: dots.ocr: Multilingual Document Layout Parsing in a Single Vision-Language Model",
   "checkpoint_or_api_version": "dots.ocr",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-12-02",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "79.1"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4947",
    "paper_arxiv_id": "2512.02498",
    "paper_id": "64224",
    "code_url": "https://github.com/rednote-hilab/dots.ocr",
    "external_source_url": null,
    "best_rank": 11,
    "original_metric": "Accuracy",
    "original_value": "79.1"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4946-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "infinity-parser-7b",
   "model": "infinity-parser-7b",
   "modelId": "infinity-parser-7b",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 82.5,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2506.03197",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for infly/Infinity-Parser-7B. Source: Infinity Parser technical report / HF leaderboard entry; metric: Accuracy.; PWC evaluation id 4946; paper: Infinity Parser: Layout Aware Reinforcement Learning for Scanned Document Parsing",
   "checkpoint_or_api_version": "Infinity-Parser-7B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-06-01",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "82.5"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4946",
    "paper_arxiv_id": "2506.03197",
    "paper_id": "50503",
    "code_url": "https://github.com/infly-ai/inf-mllm",
    "external_source_url": null,
    "best_rank": 6,
    "original_metric": "Accuracy",
    "original_value": "82.5"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4945-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "infinity-parser2-pro",
   "model": "infinity-parser2-pro",
   "modelId": "infinity-parser2-pro",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 87.6,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/infly/Infinity-Parser2-Pro",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score reported on the official Hugging Face olmOCR-bench leaderboard for infly/Infinity-Parser2-Pro. Source: Infinity Parser technical report / HF leaderboard entry; metric: Accuracy.; PWC evaluation id 4945; paper: Infinity-Parser2-Pro",
   "checkpoint_or_api_version": "Infinity-Parser2-Pro",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-05-11",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "87.6"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "4945",
    "paper_arxiv_id": null,
    "paper_id": "83158",
    "code_url": null,
    "external_source_url": "https://huggingface.co/infly/Infinity-Parser2-Pro",
    "best_rank": 1,
    "original_metric": "Accuracy",
    "original_value": "87.6"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4749-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen3-vl-235b-a22b-instruct",
   "model": "qwen3-vl-235b-a22b-instruct",
   "modelId": "qwen3-vl-235b-a22b-instruct",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 920,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2511.21631",
   "accessDate": "2026-05-14",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Table 2 of Qwen3-VL technical report (arXiv:2511.21631), OCRBench (rescaled from 87.5/92.0 to the 0-1000 scale used by the existing rows).; PWC evaluation id 4749; paper: Qwen3-VL Technical Report",
   "checkpoint_or_api_version": "Qwen3-VL-235B-A22B-Instruct",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-11-26",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "920"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "4749",
    "paper_arxiv_id": "2511.21631",
    "paper_id": "54536",
    "code_url": "https://github.com/QwenLM/Qwen3-VL",
    "external_source_url": null,
    "best_rank": 3,
    "original_metric": "Score",
    "original_value": "920"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-4748-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen3-vl-235b-a22b-thinking",
   "model": "qwen3-vl-235b-a22b-thinking",
   "modelId": "qwen3-vl-235b-a22b-thinking",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 875,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2511.21631",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Table 2 of Qwen3-VL technical report (arXiv:2511.21631), OCRBench (rescaled from 87.5/92.0 to the 0-1000 scale used by the existing rows).; PWC evaluation id 4748; paper: Qwen3-VL Technical Report",
   "checkpoint_or_api_version": "Qwen3-VL-235B-A22B-Thinking",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-11-26",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "875"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "4748",
    "paper_arxiv_id": "2511.21631",
    "paper_id": "54536",
    "code_url": "https://github.com/QwenLM/Qwen3-VL",
    "external_source_url": null,
    "best_rank": 10,
    "original_metric": "Score",
    "original_value": "875"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-3371-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "kimi-vl-a3b-thinking-2506",
   "model": "kimi-vl-a3b-thinking-2506",
   "modelId": "kimi-vl-a3b-thinking-2506",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 869,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2504.07491",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Kimi-VL-A3B-Thinking-2506 on OCRBench overall score (raw 0-1000 scale) from the moonshotai/Kimi-VL-A3B-Thinking-2506 HF model card.; PWC evaluation id 3371; paper: Kimi-VL Technical Report",
   "checkpoint_or_api_version": "Kimi-VL-A3B-Thinking-2506",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-04-10",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "869"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "3371",
    "paper_arxiv_id": "2504.07491",
    "paper_id": "47488",
    "code_url": "https://github.com/MoonshotAI/Kimi-VL",
    "external_source_url": null,
    "best_rank": 11,
    "original_metric": "Score",
    "original_value": "869"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-3351-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "kimi-vl-a3b-instruct",
   "model": "kimi-vl-a3b-instruct",
   "modelId": "kimi-vl-a3b-instruct",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 867,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2504.07491",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Kimi-VL-A3B-Instruct on OCRBench overall score (raw 0-1000 scale) from Kimi-VL Technical Report Table 3 and the moonshotai/Kimi-VL-A3B-Instruct HF model card.; PWC evaluation id 3351; paper: Kimi-VL Technical Report",
   "checkpoint_or_api_version": "Kimi-VL-A3B-Instruct",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-04-10",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "867"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "3351",
    "paper_arxiv_id": "2504.07491",
    "paper_id": "47488",
    "code_url": "https://github.com/MoonshotAI/Kimi-VL",
    "external_source_url": null,
    "best_rank": 12,
    "original_metric": "Score",
    "original_value": "867"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1327-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "chandra",
   "model": "chandra",
   "modelId": "chandra",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 83.1,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/datalab-to/chandra",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Reported in the datalab-to/chandra Hugging Face model card on olmOCR-Bench. Overall score 83.1 +/- 0.9; sub-category scores from the same table: ArXiv 82.2, Old Scans Math 80.3, Tables 88.0, Old Scans 50.4, Headers and Footers 90.8, Multi column 81.2, Long tiny text 92.3, Base 99.9. Source: own benchmarks.; PWC evaluation id 1327; paper: Chandra",
   "checkpoint_or_api_version": "Chandra",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2025-10-21",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "83.1"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "1327",
    "paper_arxiv_id": null,
    "paper_id": "83030",
    "code_url": null,
    "external_source_url": "https://huggingface.co/datalab-to/chandra",
    "best_rank": 5,
    "original_metric": "Accuracy",
    "original_value": "83.1"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1279-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen3-vl-8b-instruct",
   "model": "qwen3-vl-8b-instruct",
   "modelId": "qwen3-vl-8b-instruct",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 896,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2511.21631",
   "accessDate": "2026-05-14",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Reported on OCRBench (raw 0-1000 score) by Qwen3-VL-8B-Instruct model card.; PWC evaluation id 1279; paper: Qwen3-VL Technical Report",
   "checkpoint_or_api_version": "Qwen3-VL-8B-Instruct",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-11-26",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "896"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1279",
    "paper_arxiv_id": "2511.21631",
    "paper_id": "54536",
    "code_url": "https://github.com/QwenLM/Qwen3-VL",
    "external_source_url": null,
    "best_rank": 5,
    "original_metric": "Score",
    "original_value": "896"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1253-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "Kimi K2.5",
   "model": "Kimi K2.5",
   "modelId": "Kimi K2.5",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 88.8,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2602.02276",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; OmniDocBench v1.5 score = (1 - normalized Levenshtein distance) x 100; max 64k tokens; avg@3; Thinking mode.; PWC evaluation id 1253; paper: Kimi K2.5: Visual Agentic Intelligence",
   "checkpoint_or_api_version": "Kimi K2.5",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-02-02",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "88.8"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "1253",
    "paper_arxiv_id": "2602.02276",
    "paper_id": "55657",
    "code_url": "https://github.com/MoonshotAI/Kimi-K2.5",
    "external_source_url": null,
    "best_rank": 9,
    "original_metric": "Accuracy",
    "original_value": "88.8"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1252-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "Kimi K2.5",
   "model": "Kimi K2.5",
   "modelId": "Kimi K2.5",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 923,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2602.02276",
   "accessDate": "2026-05-13",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "OCRBench overall score on the native 0-1000 scale (card reports 92.3 normalized); max 64k tokens; avg@3; Thinking mode.; PWC evaluation id 1252; paper: Kimi K2.5: Visual Agentic Intelligence",
   "checkpoint_or_api_version": "Kimi K2.5",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-02-02",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "923"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1252",
    "paper_arxiv_id": "2602.02276",
    "paper_id": "55657",
    "code_url": "https://github.com/MoonshotAI/Kimi-K2.5",
    "external_source_url": null,
    "best_rank": 2,
    "original_metric": "Score",
    "original_value": "923"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1230-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "zaya1-vl-8b",
   "model": "zaya1-vl-8b",
   "modelId": "zaya1-vl-8b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 798,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://www.zyphra.com/zaya1-vl-8b-technical-report",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "OCRBench overall score (0-1000 scale; card reports 79.8 normalised). Reported in the ZAYA1-VL-8B technical report (Zyphra). Evaluated on the Zyphra eval harness based on VLMEvalKit.; PWC evaluation id 1230; paper: ZAYA1-VL-8B Technical Report",
   "checkpoint_or_api_version": "ZAYA1-VL-8B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-05-08",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "798"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1230",
    "paper_arxiv_id": null,
    "paper_id": "82905",
    "code_url": "https://github.com/Zyphra/transformers/tree/zaya1-vl",
    "external_source_url": null,
    "best_rank": 22,
    "original_metric": "Score",
    "original_value": "798"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1214-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "videollama3-7b",
   "model": "videollama3-7b",
   "modelId": "videollama3-7b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 828,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2501.13106",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "DAMO-NLP-SG/VideoLLaMA3-7B-Image checkpoint; numbers from the 7B-Image model card main-results table.; PWC evaluation id 1214; paper: VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding",
   "checkpoint_or_api_version": "VideoLLaMA3 7B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-01-22",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "828"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1214",
    "paper_arxiv_id": "2501.13106",
    "paper_id": "43121",
    "code_url": "https://github.com/damo-nlp-sg/videollama3",
    "external_source_url": null,
    "best_rank": 20,
    "original_metric": "Score",
    "original_value": "828"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1197-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "Qianfan-OCR",
   "model": "Qianfan-OCR",
   "modelId": "Qianfan-OCR",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 880,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.13398",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "OCRBench standard Score (0-1000); 880.; PWC evaluation id 1197; paper: Qianfan-OCR: A Unified End-to-End Model for Document Intelligence",
   "checkpoint_or_api_version": "Qianfan-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-11",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "880"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1197",
    "paper_arxiv_id": "2603.13398",
    "paper_id": "56773",
    "code_url": "https://github.com/baidubce/Qianfan-VL",
    "external_source_url": null,
    "best_rank": 7,
    "original_metric": "Score",
    "original_value": "880"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1196-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "Qianfan-OCR",
   "model": "Qianfan-OCR",
   "modelId": "Qianfan-OCR",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 79.8,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.13398",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; End-to-end OCR on olmOCR-Bench; overall 79.8 (Arxiv Math 80.1, Old Scans Math 73.1, Table Tests 81.6, Old Scans 42.0, Multi Column 80.4, Long Tiny Text 89.1, Headers Footers 92.2).; PWC evaluation id 1196; paper: Qianfan-OCR: A Unified End-to-End Model for Document Intelligence",
   "checkpoint_or_api_version": "Qianfan-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-11",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "79.8"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "1196",
    "paper_arxiv_id": "2603.13398",
    "paper_id": "56773",
    "code_url": "https://github.com/baidubce/Qianfan-VL",
    "external_source_url": null,
    "best_rank": 10,
    "original_metric": "Accuracy",
    "original_value": "79.8"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1195-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "Qianfan-OCR",
   "model": "Qianfan-OCR",
   "modelId": "Qianfan-OCR",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 93.12,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.13398",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; End-to-end document parsing on OmniDocBench v1.5; reported Overall score 93.12 (TextEdit 0.041, FormulaCDM 92.43, TableTEDs 91.02, TableTEDss 93.85, R-orderEdit 0.049). Image-to-Markdown, no external pipeline.; PWC evaluation id 1195; paper: Qianfan-OCR: A Unified End-to-End Model for Document Intelligence",
   "checkpoint_or_api_version": "Qianfan-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-11",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "93.12"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "1195",
    "paper_arxiv_id": "2603.13398",
    "paper_id": "56773",
    "code_url": "https://github.com/baidubce/Qianfan-VL",
    "external_source_url": null,
    "best_rank": 2,
    "original_metric": "Accuracy",
    "original_value": "93.12"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1178-text-edit-distance",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "minicpm-o-4-5-instruct",
   "model": "minicpm-o-4-5-instruct",
   "modelId": "minicpm-o-4-5-instruct",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.109,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2604.27393",
   "accessDate": "2026-05-13",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.0 Edit Distance; lower is better.; Instruct mode from the openbmb/MiniCPM-o-4_5 Hugging Face model card (https://huggingface.co/openbmb/MiniCPM-o-4_5); 9B params; results reported in instruct mode/variant; from the 'OmniDocBench' table; metric label in card: OverallEdit↓, EN subset (state-of-the-art for end-to-end English document parsing claim).; PWC evaluation id 1178; paper: MiniCPM-o 4.5: Towards Real-Time Full-Duplex Omni-Modal Interaction",
   "checkpoint_or_api_version": "MiniCPM-o 4.5-Instruct",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-04-30",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Edit Distance": "0.109"
   },
   "artifacts": {
    "pwc_dataset_id": "139",
    "pwc_evaluation_id": "1178",
    "paper_arxiv_id": "2604.27393",
    "paper_id": "82845",
    "code_url": "https://github.com/OpenBMB/MiniCPM-o",
    "external_source_url": null,
    "best_rank": 1,
    "original_metric": "Edit Distance",
    "original_value": "0.109"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1171-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "minicpm-o-4-5-instruct",
   "model": "minicpm-o-4-5-instruct",
   "modelId": "minicpm-o-4-5-instruct",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 876,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2604.27393",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Instruct mode from the openbmb/MiniCPM-o-4_5 Hugging Face model card (https://huggingface.co/openbmb/MiniCPM-o-4_5); 9B params; results reported in instruct mode/variant; from the 'Image Understanding (Instruct)' table; metric label in card: OCRBench; 0-1000 scale used by other rows on this leaderboard.; PWC evaluation id 1171; paper: MiniCPM-o 4.5: Towards Real-Time Full-Duplex Omni-Modal Interaction",
   "checkpoint_or_api_version": "MiniCPM-o 4.5-Instruct",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-04-30",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "876"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1171",
    "paper_arxiv_id": "2604.27393",
    "paper_id": "82845",
    "code_url": "https://github.com/OpenBMB/MiniCPM-o",
    "external_source_url": null,
    "best_rank": 9,
    "original_metric": "Score",
    "original_value": "876"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1165-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "chandra-2",
   "model": "chandra-2",
   "modelId": "chandra-2",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 85.9,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/datalab-to/chandra-ocr-2",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Chandra OCR 2 (5B params, Qwen3.5 backbone) reported on olmOCR-Bench in the HF model card; Overall score 85.9 +/- 0.8. Sub-categories on the same benchmark: ArXiv 90.2, Old Scans Math 89.3, Tables 89.9, Old Scans 49.8, Headers and Footers 92.5, Multi column 83.5, Long tiny text 92.1, Base 99.6. Source: own benchmarks, reported in the chandra-ocr-2 HF model card.; PWC evaluation id 1165; paper: Chandra OCR 2",
   "checkpoint_or_api_version": "Chandra 2",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-03-16",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "85.9"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "1165",
    "paper_arxiv_id": null,
    "paper_id": "83028",
    "code_url": "https://github.com/datalab-to/chandra",
    "external_source_url": null,
    "best_rank": 2,
    "original_metric": "Accuracy",
    "original_value": "85.9"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1157-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "falcon-ocr",
   "model": "falcon-ocr",
   "modelId": "falcon-ocr",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 88.64,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.27365",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; OmniDocBench Overall score (Accuracy %) reported by tiiuae/Falcon-OCR model card (arXiv:2603.27365, Apr 2026), aggregating three sub-metrics: Edit Distance 0.055 (lower is better), CDM 86.8 (formula recognition), TEDS 84.6 (table structure). Inference via Layout + OCR two-stage pipeline (PP-DocLayoutV3 layout detection + Falcon-OCR category-prompted VLM). Mapped to OmniDocBench v1.5 leaderboard based on overall accuracy magnitude and comparator scores.; PWC evaluation id 1157; paper: Falcon Perception",
   "checkpoint_or_api_version": "Falcon-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-28",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "88.64"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "1157",
    "paper_arxiv_id": "2603.27365",
    "paper_id": "57116",
    "code_url": "https://github.com/tiiuae/Falcon-Perception",
    "external_source_url": null,
    "best_rank": 10,
    "original_metric": "Accuracy",
    "original_value": "88.64"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1156-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "falcon-ocr",
   "model": "falcon-ocr",
   "modelId": "falcon-ocr",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 80.3,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.27365",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; olmOCR-Bench accuracy reported by tiiuae/Falcon-OCR model card (arXiv:2603.27365, Apr 2026). 'Average' across 8 category splits: ArXiv Math 80.5, Base 99.5, Headers/Footers 94.0, Long Tiny Text 78.5, Multi Column 87.1, Old Scans 43.5, Old Scans Math 69.2, Tables 90.3. Inference via Layout + OCR two-stage pipeline (PP-DocLayoutV3 layout detection + Falcon-OCR category-prompted VLM).; PWC evaluation id 1156; paper: Falcon Perception",
   "checkpoint_or_api_version": "Falcon-OCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-28",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "80.3"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "1156",
    "paper_arxiv_id": "2603.27365",
    "paper_id": "57116",
    "code_url": "https://github.com/tiiuae/Falcon-Perception",
    "external_source_url": null,
    "best_rank": 8,
    "original_metric": "Accuracy",
    "original_value": "80.3"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1153-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "dots.mocr",
   "model": "dots.mocr",
   "modelId": "dots.mocr",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 860,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.13032",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "OCRBench overall score from the dots.mocr Hugging Face model card (section 3, General Vision Tasks). The card reports 86.0 on a 0-100 scale; converted to 860 on the standard 0-1000 OCRBench scale used by other rows on this leaderboard (consistent with Qwen3-VL-2B = 85.8 -> 858 in the same table).; PWC evaluation id 1153; paper: Multimodal OCR: Parse Anything from Documents",
   "checkpoint_or_api_version": "dots.mocr",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-13",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "860"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1153",
    "paper_arxiv_id": "2603.13032",
    "paper_id": "56700",
    "code_url": "https://github.com/rednote-hilab/dots.mocr",
    "external_source_url": null,
    "best_rank": 17,
    "original_metric": "Score",
    "original_value": "860"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1148-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "dots.mocr",
   "model": "dots.mocr",
   "modelId": "dots.mocr",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 83.9,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2603.13032",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score on olmOCR-Bench from the dots.mocr Hugging Face model card (section 1.2). 3B image-text-to-text VLM; page-header and page-footer cells deleted from the result markdown per the card's note; metric source: olmocr plus internal evaluations.; PWC evaluation id 1148; paper: Multimodal OCR: Parse Anything from Documents",
   "checkpoint_or_api_version": "dots.mocr",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-03-13",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "83.9"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "1148",
    "paper_arxiv_id": "2603.13032",
    "paper_id": "56700",
    "code_url": "https://github.com/rednote-hilab/dots.mocr",
    "external_source_url": null,
    "best_rank": 3,
    "original_metric": "Accuracy",
    "original_value": "83.9"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1147-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "LightOnOCR-2-1B",
   "model": "LightOnOCR-2-1B",
   "modelId": "LightOnOCR-2-1B",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 83.2,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/lightonai/LightOnOCR-2-1B",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; Overall score on olmOCR-Bench reported on the Hugging Face model card. Score excludes the Headers/Footers (H&F) sub-test because that category rewards omission rather than transcription, while LightOnOCR-2 is trained for full-page transcription and intentionally preserves headers/footers.; PWC evaluation id 1147; paper: LightOnOCR: A 1B End-to-End Multilingual Vision-Language Model for State-of-the-Art OCR",
   "checkpoint_or_api_version": "LightOnOCR-2-1B",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-01-20",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "83.2"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "1147",
    "paper_arxiv_id": "2601.14251",
    "paper_id": "55399",
    "code_url": null,
    "external_source_url": "https://huggingface.co/lightonai/LightOnOCR-2-1B",
    "best_rank": 4,
    "original_metric": "Accuracy",
    "original_value": "83.2"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1115-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "minicpm-v-4-6-thinking-16x",
   "model": "minicpm-v-4-6-thinking-16x",
   "modelId": "minicpm-v-4-6-thinking-16x",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 831,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://huggingface.co/openbmb/MiniCPM-V-4.6",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Thinking mode from the MiniCPM-V 4.6 Hugging Face model card; official checkpoint; visual token compression ratio 16x; metric label in card: OCRBench.; PWC evaluation id 1115; paper: A Pocket-Sized MLLM for Ultra-Efficient Image and Video Understanding on Your Phone",
   "checkpoint_or_api_version": "MiniCPM-V 4.6-Thinking (16x)",
   "source_type": "vendor",
   "verification_tier": "external_registry",
   "date_run": "2026-05-13",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "831"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1115",
    "paper_arxiv_id": null,
    "paper_id": "83018",
    "code_url": null,
    "external_source_url": "https://huggingface.co/openbmb/MiniCPM-V-4.6",
    "best_rank": 19,
    "original_metric": "Score",
    "original_value": "831"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1060-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen3-5-397b-a17b",
   "model": "qwen3-5-397b-a17b",
   "modelId": "qwen3-5-397b-a17b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 931,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://qwen.ai/blog?id=qwen3.5",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Official Qwen3.5 blog (https://qwen.ai/blog?id=qwen3.5). Vision table row OCRBench; linked to OCR task using existing Score metric.; PWC evaluation id 1060; paper: Qwen3.5: Towards Native Multimodal Agents",
   "checkpoint_or_api_version": "Qwen3.5-397B-A17B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-02-16",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "931"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "1060",
    "paper_arxiv_id": null,
    "paper_id": "83017",
    "code_url": "https://github.com/huggingface/transformers",
    "external_source_url": "https://qwen.ai/blog?id=qwen3.5",
    "best_rank": 1,
    "original_metric": "Score",
    "original_value": "931"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-1059-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "qwen3-5-397b-a17b",
   "model": "qwen3-5-397b-a17b",
   "modelId": "qwen3-5-397b-a17b",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 90.8,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://qwen.ai/blog?id=qwen3.5",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; Official Qwen3.5 blog (https://qwen.ai/blog?id=qwen3.5). Vision table row OmniDocBench1.5; linked to OCR task.; PWC evaluation id 1059; paper: Qwen3.5: Towards Native Multimodal Agents",
   "checkpoint_or_api_version": "Qwen3.5-397B-A17B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2026-02-16",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "90.8"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "1059",
    "paper_arxiv_id": null,
    "paper_id": "83017",
    "code_url": "https://github.com/huggingface/transformers",
    "external_source_url": "https://qwen.ai/blog?id=qwen3.5",
    "best_rank": 6,
    "original_metric": "Accuracy",
    "original_value": "90.8"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-957-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "hunyuanocr-1b",
   "model": "hunyuanocr-1b",
   "modelId": "hunyuanocr-1b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 860,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2511.19575",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC evaluation id 957; paper: HunyuanOCR Technical Report",
   "checkpoint_or_api_version": "HunyuanOCR (1B)",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-11-24",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "860"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "957",
    "paper_arxiv_id": "2511.19575",
    "paper_id": "54349",
    "code_url": "https://github.com/Tencent-Hunyuan/HunyuanOCR",
    "external_source_url": null,
    "best_rank": 17,
    "original_metric": "Score",
    "original_value": "860"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-886-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "minimax-vl-01",
   "model": "minimax-vl-01",
   "modelId": "minimax-vl-01",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 865,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2501.08313",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC evaluation id 886; paper: MiniMax-01: Scaling Foundation Models with Lightning Attention",
   "checkpoint_or_api_version": "MiniMax-VL-01",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-01-14",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "865"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "886",
    "paper_arxiv_id": "2501.08313",
    "paper_id": "42842",
    "code_url": "https://github.com/MiniMax-AI",
    "external_source_url": null,
    "best_rank": 14,
    "original_metric": "Score",
    "original_value": "865"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-768-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "internvl3-78b",
   "model": "internvl3-78b",
   "modelId": "internvl3-78b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 906,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2504.10479",
   "accessDate": "2026-05-14",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC evaluation id 768; paper: InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models",
   "checkpoint_or_api_version": "InternVL3-78B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-04-14",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "906"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "768",
    "paper_arxiv_id": "2504.10479",
    "paper_id": "47682",
    "code_url": "https://github.com/opengvlab/internvl",
    "external_source_url": null,
    "best_rank": 4,
    "original_metric": "Score",
    "original_value": "906"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-681-text-edit-distance",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "deepseek-ocr-gundam-m",
   "model": "deepseek-ocr-gundam-m",
   "modelId": "deepseek-ocr-gundam-m",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "text-edit-distance",
   "value": 0.123,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2510.18234",
   "accessDate": "2026-05-13",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.0 Edit Distance; lower is better.; PWC evaluation id 681; paper: DeepSeek-OCR: Contexts Optical Compression",
   "checkpoint_or_api_version": "DeepSeek-OCR (Gundam-M)",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-10-21",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Edit Distance": "0.123"
   },
   "artifacts": {
    "pwc_dataset_id": "139",
    "pwc_evaluation_id": "681",
    "paper_arxiv_id": "2510.18234",
    "paper_id": "53719",
    "code_url": "https://github.com/deepseek-ai/DeepSeek-OCR",
    "external_source_url": null,
    "best_rank": 2,
    "original_metric": "Edit Distance",
    "original_value": "0.123"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-680-precision",
   "task_id": "ocr",
   "benchmark_id": "fox-english-subset-600-1300-text-tokens",
   "model_id": "deepseek-ocr-small-100-vision-tokens",
   "model": "deepseek-ocr-small-100-vision-tokens",
   "modelId": "deepseek-ocr-small-100-vision-tokens",
   "dataset": "fox-english-subset-600-1300-text-tokens",
   "datasetId": "fox-english-subset-600-1300-text-tokens",
   "metric": "precision",
   "value": 98.5,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2510.18234",
   "accessDate": "2025-11-11",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC Fox English OCR subset Precision.; PWC evaluation id 680; paper: DeepSeek-OCR: Contexts Optical Compression",
   "checkpoint_or_api_version": "DeepSeek-OCR (Small, 100 vision tokens)",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-10-21",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Precision": "98.5"
   },
   "artifacts": {
    "pwc_dataset_id": "313",
    "pwc_evaluation_id": "680",
    "paper_arxiv_id": "2510.18234",
    "paper_id": "53719",
    "code_url": "https://github.com/deepseek-ai/DeepSeek-OCR",
    "external_source_url": null,
    "best_rank": 1,
    "original_metric": "Precision",
    "original_value": "98.5"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-679-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "olmocr-2-7b-1025-7b",
   "model": "olmocr-2-7b-1025-7b",
   "modelId": "olmocr-2-7b-1025-7b",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 82.4,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2510.19817",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; PWC evaluation id 679; paper: olmOCR 2: Unit Test Rewards for Document OCR",
   "checkpoint_or_api_version": "olmOCR-2-7B-1025 (7B)",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-10-22",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "82.4"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "679",
    "paper_arxiv_id": "2510.19817",
    "paper_id": "53756",
    "code_url": null,
    "external_source_url": null,
    "best_rank": 7,
    "original_metric": "Accuracy",
    "original_value": "82.4"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-145-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen2-vl-2b",
   "model": "qwen2-vl-2b",
   "modelId": "qwen2-vl-2b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 809,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2409.12191",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC evaluation id 145; paper: Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution",
   "checkpoint_or_api_version": "Qwen2-VL-2B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2024-09-18",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "809"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "145",
    "paper_arxiv_id": "2409.12191",
    "paper_id": "37055",
    "code_url": "https://github.com/qwenlm/qwen2-vl",
    "external_source_url": null,
    "best_rank": 21,
    "original_metric": "Score",
    "original_value": "809"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-144-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "qwen2-vl-7b",
   "model": "qwen2-vl-7b",
   "modelId": "qwen2-vl-7b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 866,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2409.12191",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC evaluation id 144; paper: Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution",
   "checkpoint_or_api_version": "Qwen2-VL-7B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2024-09-18",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "866"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "144",
    "paper_arxiv_id": "2409.12191",
    "paper_id": "37055",
    "code_url": "https://github.com/qwenlm/qwen2-vl",
    "external_source_url": null,
    "best_rank": 13,
    "original_metric": "Score",
    "original_value": "866"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-143-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "Qwen2-VL 72B",
   "model": "Qwen2-VL 72B",
   "modelId": "Qwen2-VL 72B",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 877,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2409.12191",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC evaluation id 143; paper: Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution",
   "checkpoint_or_api_version": "Qwen2-VL-72B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2024-09-18",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "877"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "143",
    "paper_arxiv_id": "2409.12191",
    "paper_id": "37055",
    "code_url": "https://github.com/qwenlm/qwen2-vl",
    "external_source_url": null,
    "best_rank": 8,
    "original_metric": "Score",
    "original_value": "877"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-115-score",
   "task_id": "ocr",
   "benchmark_id": "ocrbench",
   "model_id": "videollama3-2b",
   "model": "videollama3-2b",
   "modelId": "videollama3-2b",
   "dataset": "ocrbench",
   "datasetId": "ocrbench",
   "metric": "score",
   "value": 779,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2501.13106",
   "accessDate": "2026-05-18",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "PWC evaluation id 115; paper: VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding",
   "checkpoint_or_api_version": "VideoLLaMA3 2B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-01-22",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Score": "779"
   },
   "artifacts": {
    "pwc_dataset_id": "155",
    "pwc_evaluation_id": "115",
    "paper_arxiv_id": "2501.13106",
    "paper_id": "43121",
    "code_url": "https://github.com/damo-nlp-sg/videollama3",
    "external_source_url": null,
    "best_rank": 24,
    "original_metric": "Score",
    "original_value": "779"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-93-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "mineru-2.5",
   "model": "mineru-2.5",
   "modelId": "mineru-2.5",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 77.5,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2509.22186",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; PWC evaluation id 93; paper: MinerU2.5: A Decoupled Vision-Language Model for Efficient\n  High-Resolution Document Parsing",
   "checkpoint_or_api_version": "MinerU2.5",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-09-26",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "77.5"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "93",
    "paper_arxiv_id": "2509.22186",
    "paper_id": "52956",
    "code_url": "https://github.com/opendatalab/MinerU",
    "external_source_url": null,
    "best_rank": 12,
    "original_metric": "Accuracy",
    "original_value": "77.5"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-92-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "paddleocr-vl",
   "model": "paddleocr-vl",
   "modelId": "paddleocr-vl",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 92.56,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2510.14528",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; PWC evaluation id 92; paper: PaddleOCR-VL: Boosting Multilingual Document Parsing via a 0.9B\n  Ultra-Compact Vision-Language Model",
   "checkpoint_or_api_version": "PaddleOCR-VL",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-10-16",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "92.56"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "92",
    "paper_arxiv_id": "2510.14528",
    "paper_id": "53603",
    "code_url": "https://github.com/PaddlePaddle/PaddleOCR",
    "external_source_url": null,
    "best_rank": 4,
    "original_metric": "Accuracy",
    "original_value": "92.56"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-91-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "olmocr",
   "model": "olmocr",
   "modelId": "olmocr",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 81.79,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2502.18443",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; PWC evaluation id 91; paper: olmOCR: Unlocking Trillions of Tokens in PDFs with Vision Language Models",
   "checkpoint_or_api_version": "olmOCR",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-02-25",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "81.79"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "91",
    "paper_arxiv_id": "2502.18443",
    "paper_id": "45019",
    "code_url": "https://github.com/allenai/olmocr",
    "external_source_url": null,
    "best_rank": 13,
    "original_metric": "Accuracy",
    "original_value": "81.79"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-90-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "MonkeyOCR-pro-1.2B",
   "model": "MonkeyOCR-pro-1.2B",
   "modelId": "MonkeyOCR-pro-1.2B",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 86.96,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2506.05218",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; PWC evaluation id 90; paper: MonkeyOCR: Document Parsing with a Structure-Recognition-Relation Triplet Paradigm",
   "checkpoint_or_api_version": "MonkeyOCR-pro-1.2B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-06-05",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "86.96"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "90",
    "paper_arxiv_id": "2506.05218",
    "paper_id": "50633",
    "code_url": "https://github.com/yuliang-liu/monkeyocr",
    "external_source_url": null,
    "best_rank": 12,
    "original_metric": "Accuracy",
    "original_value": "86.96"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-89-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "MonkeyOCR-3B",
   "model": "MonkeyOCR-3B",
   "modelId": "MonkeyOCR-3B",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 87.13,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2506.05218",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; PWC evaluation id 89; paper: MonkeyOCR: Document Parsing with a Structure-Recognition-Relation Triplet Paradigm",
   "checkpoint_or_api_version": "MonkeyOCR-3B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-06-05",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "87.13"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "89",
    "paper_arxiv_id": "2506.05218",
    "paper_id": "50633",
    "code_url": "https://github.com/yuliang-liu/monkeyocr",
    "external_source_url": null,
    "best_rank": 11,
    "original_metric": "Accuracy",
    "original_value": "87.13"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-88-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "MonkeyOCR-pro-3B",
   "model": "MonkeyOCR-pro-3B",
   "modelId": "MonkeyOCR-pro-3B",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 88.85,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2506.05218",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; PWC evaluation id 88; paper: MonkeyOCR: Document Parsing with a Structure-Recognition-Relation Triplet Paradigm",
   "checkpoint_or_api_version": "MonkeyOCR-pro-3B",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-06-05",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "88.85"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "88",
    "paper_arxiv_id": "2506.05218",
    "paper_id": "50633",
    "code_url": "https://github.com/yuliang-liu/monkeyocr",
    "external_source_url": null,
    "best_rank": 8,
    "original_metric": "Accuracy",
    "original_value": "88.85"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-87-composite",
   "task_id": "ocr",
   "benchmark_id": "omnidocbench",
   "model_id": "mineru-2.5",
   "model": "mineru-2.5",
   "modelId": "mineru-2.5",
   "dataset": "omnidocbench",
   "datasetId": "omnidocbench",
   "metric": "composite",
   "value": 90.67,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2509.22186",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC OmniDocBench v1.5 Accuracy.; PWC evaluation id 87; paper: MinerU2.5: A Decoupled Vision-Language Model for Efficient\n  High-Resolution Document Parsing",
   "checkpoint_or_api_version": "MinerU2.5",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-09-26",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "90.67"
   },
   "artifacts": {
    "pwc_dataset_id": "140",
    "pwc_evaluation_id": "87",
    "paper_arxiv_id": "2509.22186",
    "paper_id": "52956",
    "code_url": "https://github.com/opendatalab/MinerU",
    "external_source_url": null,
    "best_rank": 7,
    "original_metric": "Accuracy",
    "original_value": "90.67"
   }
  },
  {
   "schema_version": "1.0",
   "result_id": "pwc-86-pass-rate",
   "task_id": "ocr",
   "benchmark_id": "olmocr-bench",
   "model_id": "paddleocr-vl",
   "model": "paddleocr-vl",
   "modelId": "paddleocr-vl",
   "dataset": "olmocr-bench",
   "datasetId": "olmocr-bench",
   "metric": "pass-rate",
   "value": 80,
   "source": "paperswithcode-public-api",
   "sourceUrl": "https://arxiv.org/abs/2510.14528",
   "accessDate": "2026-05-15",
   "verified": true,
   "verifiedBy": "Papers With Code public API dump",
   "notes": "Mapped from PWC olmOCR-Bench Accuracy.; PWC evaluation id 86; paper: PaddleOCR-VL: Boosting Multilingual Document Parsing via a 0.9B\n  Ultra-Compact Vision-Language Model",
   "checkpoint_or_api_version": "PaddleOCR-VL",
   "source_type": "paper",
   "verification_tier": "external_registry",
   "date_run": "2025-10-16",
   "sample_count": null,
   "eval_set_hash": null,
   "hardware": null,
   "inference_engine": null,
   "streaming": null,
   "metrics": {
    "Accuracy": "80.0"
   },
   "artifacts": {
    "pwc_dataset_id": "141",
    "pwc_evaluation_id": "86",
    "paper_arxiv_id": "2510.14528",
    "paper_id": "53603",
    "code_url": "https://github.com/PaddlePaddle/PaddleOCR",
    "external_source_url": null,
    "best_rank": 9,
    "original_metric": "Accuracy",
    "original_value": "80.0"
   }
  },
  {
   "modelId": "",
   "datasetId": "omnidocbench",
   "accessDate": "2026-06-16",
   "model": "PaddleOCR-VL-1.6",
   "dataset": "omnidocbench",
   "metric": "composite",
   "value": 96.33,
   "source": "vendor model card",
   "sourceUrl": "https://huggingface.co/PaddlePaddle/PaddleOCR-VL-1.6",
   "verified": false,
   "verifiedBy": "deep-research-2026-06",
   "notes": "OmniDocBench v1.6, vendor self-reported (arXiv 2606.03264)"
  },
  {
   "modelId": "",
   "datasetId": "omnidocbench",
   "accessDate": "2026-06-16",
   "model": "MinerU2.5-Pro",
   "dataset": "omnidocbench",
   "metric": "composite",
   "value": 95.69,
   "source": "paper",
   "sourceUrl": "https://arxiv.org/abs/2604.04771",
   "verified": false,
   "verifiedBy": "deep-research-2026-06",
   "notes": "OmniDocBench v1.6, vendor (arXiv 2604.04771)"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "KDL Frontier",
   "dataset": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 68.1,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 EN #1, official leaderboard 2026.03 (closed)"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "Nemotron-3-Nano-Omni-30B",
   "dataset": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 65.8,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 EN top open model, official leaderboard 2026.03"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "Gemini 3 Pro Preview",
   "dataset": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 63.4,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 EN, official leaderboard 2026.03"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "TeleMM-2.0",
   "dataset": "ocrbench-v2",
   "metric": "overall-en-private",
   "value": 61.8,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 EN, official leaderboard 2026.03 (closed)"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "TeleMM-2.0",
   "dataset": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 66.2,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 ZH #1, official leaderboard 2026.03 (closed)"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "Qwen3.5-9B",
   "dataset": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 64.1,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 ZH top open model, official leaderboard 2026.03"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "Gemini 3 Pro Preview",
   "dataset": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 63.8,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 ZH, official leaderboard 2026.03"
  },
  {
   "modelId": "",
   "datasetId": "ocrbench-v2",
   "accessDate": "2026-06-16",
   "model": "MiniCPM-o-4.5",
   "dataset": "ocrbench-v2",
   "metric": "overall-zh-private",
   "value": 61.5,
   "source": "OCRBench v2 official leaderboard",
   "sourceUrl": "https://99franklin.github.io/ocrbench_v2/",
   "verified": true,
   "verifiedBy": "official-leaderboard-2026.03",
   "notes": "OCRBench v2 ZH, official leaderboard 2026.03"
  },
  {
   "modelId": "",
   "datasetId": "olmocr-bench",
   "accessDate": "2026-06-23",
   "model": "surya-2",
   "dataset": "olmocr-bench",
   "metric": "pass-rate",
   "value": 83.3,
   "source": "vendor model card",
   "sourceUrl": "https://github.com/datalab-to/surya",
   "verified": false,
   "verifiedBy": "deep-research-2026-06",
   "notes": "olmOCR-Bench, vendor self-reported; Surya OCR 2 v0.20.0, 650M, 90+ languages"
  },
  {
   "modelId": "",
   "datasetId": "omnidocbench",
   "accessDate": "2026-06-23",
   "model": "unlimited-ocr",
   "dataset": "omnidocbench",
   "metric": "composite",
   "value": 93.92,
   "source": "vendor model card",
   "sourceUrl": "https://huggingface.co/baidu/Unlimited-OCR",
   "verified": false,
   "verifiedBy": "deep-research-2026-06",
   "notes": "OmniDocBench v1.6, vendor self-reported (arXiv 2606.23050); 3B MoE ~500M active, long-horizon 40+ page single-pass parsing built on DeepSeek-OCR"
  }
 ],
 "speedBenchmarks": [],
 "pendingVerification": [
  {
   "model": "trocr-large",
   "dataset": "sroie",
   "metric": "f1",
   "claimedValue": 96.58,
   "source": "arxiv-paper",
   "sourceUrl": "https://arxiv.org/abs/2109.10282",
   "status": "needs-pdf-verification",
   "notes": "TrOCR paper claims SOTA on SROIE. Exact number needs verification from PDF."
  },
  {
   "model": "trocr-large",
   "dataset": "iam",
   "metric": "cer",
   "claimedValue": 2.89,
   "source": "arxiv-paper",
   "sourceUrl": "https://arxiv.org/abs/2109.10282",
   "status": "needs-pdf-verification",
   "notes": "TrOCR paper claims SOTA on IAM handwriting. Exact number needs verification from PDF."
  },
  {
   "model": "paddleocr-v4",
   "dataset": "icdar-2015",
   "metric": "f1",
   "claimedValue": null,
   "source": "github-readme",
   "sourceUrl": "https://github.com/PaddlePaddle/PaddleOCR",
   "status": "needs-documentation-verification",
   "notes": "PaddleOCR claims top performance but exact ICDAR 2015 numbers not found in README. Check benchmark docs."
  },
  {
   "model": "polish-roberta-ocr",
   "dataset": "poleval-2021-ocr",
   "metric": "cer",
   "value": 2.1,
   "source": "poleval-2021",
   "sourceUrl": "http://2021.poleval.pl/tasks/task1",
   "accessDate": "2025-12-19",
   "notes": "Winning solution at PolEval 2021 OCR Correction Task. Uses Polish RoBERTa for sequence-to-sequence correction."
  },
  {
   "model": "polish-t5-ocr",
   "dataset": "poleval-2021-ocr",
   "metric": "cer",
   "value": 2.4,
   "source": "poleval-2021",
   "sourceUrl": "http://2021.poleval.pl/tasks/task1",
   "accessDate": "2025-12-19",
   "notes": "Polish T5-based OCR correction model. Second place at PolEval 2021."
  },
  {
   "model": "herbert",
   "dataset": "poleval-2021-ocr",
   "metric": "cer",
   "value": 2.8,
   "source": "poleval-2021",
   "sourceUrl": "http://2021.poleval.pl/tasks/task1",
   "accessDate": "2025-12-19",
   "notes": "HerBERT-based OCR correction approach. Strong baseline for Polish OCR."
  },
  {
   "model": "abbyy-finereader",
   "dataset": "impact-psnc",
   "metric": "cer",
   "value": 1.2,
   "source": "impact-project",
   "sourceUrl": "https://dl.psnc.pl/activities/projekty/impact/",
   "accessDate": "2025-12-19",
   "notes": "ABBYY FineReader performance on Polish historical documents. Best on antiqua fonts."
  },
  {
   "model": "tesseract-polish",
   "dataset": "impact-psnc",
   "metric": "cer",
   "value": 3.8,
   "source": "impact-project",
   "sourceUrl": "https://dl.psnc.pl/activities/projekty/impact/",
   "accessDate": "2025-12-19",
   "notes": "Tesseract with Polish language model on IMPACT-PSNC benchmark. Lower accuracy on gothic fonts."
  },
  {
   "model": "abbyy-finereader",
   "dataset": "impact-psnc",
   "metric": "word-accuracy",
   "value": 97.5,
   "source": "impact-project",
   "sourceUrl": "https://dl.psnc.pl/activities/projekty/impact/",
   "accessDate": "2025-12-19",
   "notes": "Word-level accuracy on Polish historical documents from four digital libraries."
  },
  {
   "model": "tesseract-polish",
   "dataset": "impact-psnc",
   "metric": "word-accuracy",
   "value": 92.1,
   "source": "impact-project",
   "sourceUrl": "https://dl.psnc.pl/activities/projekty/impact/",
   "accessDate": "2025-12-19",
   "notes": "Tesseract word-level accuracy. Struggles with Polish diacritics in gothic fonts."
  },
  {
   "model": "tesseract-polish",
   "dataset": "codesota-polish",
   "metric": "cer",
   "value": 26.3,
   "source": "codesota",
   "sourceUrl": "https://codesota.com/polish-ocr",
   "accessDate": "2025-12-20",
   "notes": "Tesseract 5.5.1 baseline on CodeSOTA Polish benchmark (1000 images). Overall CER across all 4 categories and 5 degradation levels."
  },
  {
   "model": "tesseract-polish",
   "dataset": "codesota-polish",
   "metric": "wer",
   "value": 38.5,
   "source": "codesota",
   "sourceUrl": "https://codesota.com/polish-ocr",
   "accessDate": "2025-12-20",
   "notes": "Tesseract 5.5.1 Word Error Rate. Higher than CER due to diacritic confusion causing whole-word failures."
  },
  {
   "model": "tesseract-polish",
   "dataset": "codesota-polish",
   "metric": "accuracy",
   "value": 73.7,
   "source": "codesota",
   "sourceUrl": "https://codesota.com/polish-ocr",
   "accessDate": "2025-12-20",
   "notes": "Character-level accuracy (100% - CER). Synthetic text categories drag down the overall score."
  },
  {
   "model": "tesseract-polish",
   "dataset": "codesota-polish-wikipedia",
   "metric": "cer",
   "value": 5.2,
   "source": "codesota",
   "sourceUrl": "https://codesota.com/polish-ocr",
   "accessDate": "2025-12-20",
   "notes": "Best category - potential data contamination from training on Wikipedia text. CER 94.8% accuracy."
  },
  {
   "model": "tesseract-polish",
   "dataset": "codesota-polish-real",
   "metric": "cer",
   "value": 7.3,
   "source": "codesota",
   "sourceUrl": "https://codesota.com/polish-ocr",
   "accessDate": "2025-12-20",
   "notes": "Real Polish corpus (Pan Tadeusz, official documents). CER 92.7% accuracy."
  },
  {
   "model": "tesseract-polish",
   "dataset": "codesota-polish-synth-random",
   "metric": "cer",
   "value": 40.6,
   "source": "codesota",
   "sourceUrl": "https://codesota.com/polish-ocr",
   "accessDate": "2025-12-20",
   "notes": "Random Polish character sequences - no language model assistance. Tests pure character recognition."
  },
  {
   "model": "tesseract-polish",
   "dataset": "codesota-polish-synth-words",
   "metric": "cer",
   "value": 52.1,
   "source": "codesota",
   "sourceUrl": "https://codesota.com/polish-ocr",
   "accessDate": "2025-12-20",
   "notes": "Markov-generated Polish-like words. No dictionary fallback possible - worst category for Tesseract."
  },
  {
   "model": "claude-sonnet-4",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 77.2,
   "source": "anthropic",
   "sourceUrl": "https://www.anthropic.com/research/swe-bench-sonnet",
   "accessDate": "2025-12-24",
   "notes": "10 trials averaged, no test-time compute, 200K thinking budget on full 500-problem set."
  },
  {
   "model": "claude-sonnet-4-high-compute",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 82,
   "source": "anthropic",
   "sourceUrl": "https://www.anthropic.com/research/swe-bench-sonnet",
   "accessDate": "2025-12-24",
   "notes": "High compute configuration."
  },
  {
   "model": "claude-opus-4.5",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 74.6,
   "source": "llm-stats",
   "sourceUrl": "https://llm-stats.com/benchmarks/swe-bench-verified",
   "accessDate": "2025-12-24",
   "notes": "Non-thinking mode. NEEDS_BETTER_SOURCE: llm-stats is not primary — replace with Anthropic official.",
   "needsBetterSource": true
  },
  {
   "model": "o3",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 72,
   "source": "openai",
   "sourceUrl": "https://openai.com/index/introducing-swe-bench-verified/",
   "accessDate": "2025-12-24",
   "notes": "OpenAI o3 reasoning model."
  },
  {
   "model": "claude-3.7-sonnet",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 70.3,
   "source": "anthropic",
   "sourceUrl": "https://www.anthropic.com/news/claude-sonnet-4-5",
   "accessDate": "2025-12-24",
   "notes": "With custom scaffold."
  },
  {
   "model": "claude-3.5-sonnet",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 49,
   "source": "anthropic",
   "sourceUrl": "https://www.anthropic.com/news/claude-sonnet-4-5",
   "accessDate": "2025-12-24",
   "notes": "October 2024 upgraded version."
  },
  {
   "model": "o1",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 48.9,
   "source": "openai",
   "sourceUrl": "https://openai.com/index/introducing-swe-bench-verified/",
   "accessDate": "2025-12-24",
   "notes": "OpenAI o1 reasoning model."
  },
  {
   "model": "gpt-4o",
   "dataset": "swe-bench-verified",
   "metric": "accuracy",
   "value": 33.2,
   "source": "openai",
   "sourceUrl": "https://openai.com/index/introducing-swe-bench-verified/",
   "accessDate": "2025-12-24",
   "notes": "Best performing scaffold, August 2024."
  },
  {
   "model": "o3",
   "dataset": "aime-2024",
   "metric": "accuracy",
   "value": 96.7,
   "source": "openai",
   "sourceUrl": "https://www.thealgorithmicbridge.com/p/openai-o3-model-is-a-message-from",
   "accessDate": "2025-12-24",
   "notes": "OpenAI o3 on AIME 2024 math olympiad qualifying exam."
  },
  {
   "model": "o1",
   "dataset": "aime-2024",
   "metric": "accuracy",
   "value": 83.3,
   "source": "openai",
   "sourceUrl": "https://www.thealgorithmicbridge.com/p/openai-o3-model-is-a-message-from",
   "accessDate": "2025-12-24",
   "notes": "OpenAI o1 reasoning model."
  },
  {
   "model": "deepseek-r1",
   "dataset": "aime-2024",
   "metric": "accuracy",
   "value": 79.8,
   "source": "vals-ai",
   "sourceUrl": "https://www.vals.ai/benchmarks/aime-2025-03-11",
   "accessDate": "2025-12-24",
   "notes": "DeepSeek R1 open-source reasoning model."
  },
  {
   "model": "o1",
   "dataset": "aime-2024",
   "metric": "accuracy",
   "value": 74.4,
   "source": "vellum",
   "sourceUrl": "https://www.vellum.ai/blog/analysis-openai-o1-vs-gpt-4o",
   "accessDate": "2025-12-24",
   "notes": "International Mathematical Olympiad qualifying exam."
  },
  {
   "model": "gpt-4o",
   "dataset": "aime-2024",
   "metric": "accuracy",
   "value": 9.3,
   "source": "vellum",
   "sourceUrl": "https://www.vellum.ai/blog/analysis-openai-o1-vs-gpt-4o",
   "accessDate": "2025-12-24",
   "notes": "Non-reasoning model baseline."
  },
  {
   "model": "o3",
   "dataset": "gpqa-diamond",
   "metric": "accuracy",
   "value": 87.7,
   "source": "epoch-ai",
   "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
   "accessDate": "2025-12-24",
   "notes": "Well above PhD expert performance (65%)."
  },
  {
   "model": "gemini-2.5-pro",
   "dataset": "gpqa-diamond",
   "metric": "accuracy",
   "value": 86.4,
   "source": "epoch-ai",
   "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
   "accessDate": "2025-12-24",
   "notes": "Google Gemini 2.5 Pro."
  },
  {
   "model": "o1",
   "dataset": "gpqa-diamond",
   "metric": "accuracy",
   "value": 77.3,
   "source": "openai",
   "sourceUrl": "https://klu.ai/glossary/gpqa-eval",
   "accessDate": "2025-12-24",
   "notes": "Zero-shot pass@1. First model to surpass PhD expert baseline."
  },
  {
   "model": "o3-mini",
   "dataset": "gpqa-diamond",
   "metric": "accuracy",
   "value": 75,
   "source": "epoch-ai",
   "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
   "accessDate": "2025-12-24",
   "notes": "Average accuracy."
  },
  {
   "model": "claude-3.5-sonnet",
   "dataset": "gpqa-diamond",
   "metric": "accuracy",
   "value": 59.4,
   "source": "epoch-ai",
   "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
   "accessDate": "2025-12-24",
   "notes": "Zero-shot Chain-of-Thought, June 2024."
  },
  {
   "model": "gpt-4o",
   "dataset": "gpqa-diamond",
   "metric": "accuracy",
   "value": 50.6,
   "source": "epoch-ai",
   "sourceUrl": "https://epoch.ai/benchmarks/gpqa-diamond",
   "accessDate": "2025-12-24",
   "notes": "Non-reasoning model."
  }
 ]
}
