{
  "updated": "2026-07-20",
  "latest_generation": "7-2026",
  "benchmarks": [
    {
      "id": "bench-effortless-7-2026",
      "label": "Effortless",
      "tier": 1,
      "generation": "7-2026",
      "rows": 300,
      "metric": "exact_match",
      "dual_mode": true,
      "dataset_url": "https://huggingface.co/datasets/bench-labs/bench-effortless-7-2026",
      "description": "Sanity-layer benchmark, v2 dual-mode schema: every item scored by generation AND log-likelihood. Balanced across all 17 categories."
    },
    {
      "id": "bench-easy-7-2026",
      "label": "Easy",
      "tier": 2,
      "generation": "7-2026",
      "rows": 300,
      "metric": "hybrid_score",
      "dual_mode": true,
      "dataset_url": "https://huggingface.co/datasets/bench-labs/bench-easy-7-2026",
      "description": "Easy-tier QA, v2 dual-mode schema with per-item scorer routing (strict / semantic / fuzzy). Balanced across all 17 categories."
    },
    {
      "id": "bench-mid-7-2026",
      "label": "Mid",
      "tier": 3,
      "generation": "7-2026",
      "rows": 300,
      "metric": "soft_score_norm",
      "dual_mode": true,
      "dataset_url": "https://huggingface.co/datasets/bench-labs/bench-mid-7-2026",
      "description": "Mid-tier QA, v2 dual-mode schema; headline is log-likelihood soft_score_norm, generation metrics also reported. Balanced across all 17 categories."
    },
    {
      "id": "bench-effortless-6-2026",
      "label": "Effortless (6-2026)",
      "tier": 1,
      "generation": "6-2026",
      "rows": 240,
      "metric": "exact_match",
      "dataset_url": "https://huggingface.co/datasets/bench-labs/bench-effortless-6-2026",
      "description": "Sanity-layer benchmark. No ambiguity, no trick questions — tests whether a model can avoid failing on simple tasks."
    },
    {
      "id": "bench-easy-6-2026",
      "label": "Easy (6-2026)",
      "tier": 2,
      "generation": "6-2026",
      "rows": 238,
      "metric": "hybrid_score",
      "dataset_url": "https://huggingface.co/datasets/bench-labs/bench-easy-6-2026",
      "description": "Effortless-to-Easy tier QA. Structured language understanding and controlled generalization."
    },
    {
      "id": "bench-mid-6-2026",
      "label": "Mid (6-2026)",
      "tier": 3,
      "generation": "6-2026",
      "rows": 143,
      "metric": "soft_score_norm",
      "dataset_url": "https://huggingface.co/datasets/bench-labs/bench-mid-6-2026",
      "description": "Easy-to-Mid tier multiple-choice QA scored via lm-eval loglikelihood, target_scores dict."
    },
    {
      "id": "bench-AGI",
      "label": "AGI",
      "tier": 4,
      "generation": "6-2026",
      "rows": null,
      "metric": "rank_order",
      "dataset_url": "https://huggingface.co/datasets/bench-labs/bench-AGI",
      "description": "Hard, open-ended questions with human-argued reasoning traces, graded blind by a multi-vendor panel. Rank-order evaluation, not absolute scoring.",
      "unavailable": true,
      "unavailable_reason": "Under maintenance. Scoring pipeline is being reworked — see the dataset README."
    }
  ],
  "categories": [
    "Knowledge-basic",
    "Knowledge-definitions",
    "Math-arithmetic",
    "Math-pattern",
    "Math-reasoning",
    "Logic-deduction",
    "Logic-pattern",
    "Logic-consistency",
    "Language-comprehension",
    "Language-transformation",
    "Language-structure",
    "Commonsense-reasoning",
    "Commonsense-causality",
    "Commonsense-simulation",
    "Pattern-matching",
    "Pattern-recognition",
    "Pattern-generation"
  ],
  "models": [
    {
      "id": "liquidai-lfm2.5-1.2b-base",
      "name": "LiquidAI/LFM2.5-1.2B-Base",
      "org": "LiquidAI",
      "params_b": null,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/LiquidAI/LFM2.5-1.2B-Base",
      "model_revision": "f6a5d174bc3e52bd0df245d69133f9930b4828d8",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.0,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.0
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.0
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.0
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.0
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.3087,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.7338
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.7347
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.742
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.0303
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.8057
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.7448
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.2311
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.7685
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0667
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.0868
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.0
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.621,
          "n": 143,
          "acc": 0.5385,
          "acc_norm": 0.6154,
          "soft_score": 0.5455,
          "soft_score_norm": 0.621,
          "stderr": 0.0404,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 0.6,
              "acc_norm": 0.6
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.5
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.7
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 0.5714,
              "acc_norm": 0.5714
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.9
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.7
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.4
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.7
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.3
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.6
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 0.875,
              "acc_norm": 0.875
            },
            "Math-pattern": {
              "n": 7,
              "acc": 0.8571,
              "acc_norm": 0.8571
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.3
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 0.875
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.4444,
              "acc_norm": 0.6667
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "liquidai-lfm2.5-1.2b-instruct",
      "name": "LiquidAI/LFM2.5-1.2B-Instruct",
      "org": "LiquidAI",
      "params_b": null,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/LiquidAI/LFM2.5-1.2B-Instruct",
      "model_revision": "868df74dd56ff8a0c2ac5dbf281690c2dbebe4c9",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.1208,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.119
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.0
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0952
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.475
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.027
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.441,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.7508
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.7711
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.7765
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.303
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.8172
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.7459
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.4632
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.7849
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0667
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.2
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.2727
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.4654
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.2
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.607,
          "n": 143,
          "acc": 0.5175,
          "acc_norm": 0.6014,
          "soft_score": 0.5231,
          "soft_score_norm": 0.607,
          "stderr": 0.0406,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 0.8,
              "acc_norm": 0.8
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.5
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.6
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.9
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.6
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.2,
              "acc_norm": 0.3
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.6
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.2
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.0,
              "acc_norm": 0.3
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.3
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Math-pattern": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.3
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.75
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 0.75,
              "acc_norm": 0.75
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.5556,
              "acc_norm": 0.6667
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "liquidai-lfm2.5-230m-base",
      "name": "LiquidAI/LFM2.5-230M-Base",
      "org": "LiquidAI",
      "params_b": null,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/LiquidAI/LFM2.5-230M-Base",
      "model_revision": "9d2be5519834990d30996f878b6771cccbd24f2c",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.0042,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.0
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.0
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.0
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.027
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.2876,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.7301
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.7137
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.7434
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.7888
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.7182
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.2974
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.6079
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.0247
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.0
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.5301,
          "n": 143,
          "acc": 0.4266,
          "acc_norm": 0.5245,
          "soft_score": 0.4357,
          "soft_score_norm": 0.5301,
          "stderr": 0.0415,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 0.8,
              "acc_norm": 0.6
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.4
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.4
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 0.5714,
              "acc_norm": 0.8571
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.1,
              "acc_norm": 0.7
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.8
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.1,
              "acc_norm": 0.3
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.2,
              "acc_norm": 0.4
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.7
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.4
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 0.875,
              "acc_norm": 0.875
            },
            "Math-pattern": {
              "n": 7,
              "acc": 0.8571,
              "acc_norm": 0.8571
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.3
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.5
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 0.5,
              "acc_norm": 0.5
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.3333,
              "acc_norm": 0.3333
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "liquidai-lfm2.5-350m",
      "name": "LiquidAI/LFM2.5-350M",
      "org": "LiquidAI",
      "params_b": null,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/LiquidAI/LFM2.5-350M",
      "model_revision": "b9d6e4e2d75f440b12a2b4d731c808004ecbbd89",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.1375,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.027
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.0952
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.2143
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.4286
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.0
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.027
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.358,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.721
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.7313
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.75
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.2424
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.7996
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.741
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.3121
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.7689
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.1
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.2
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.1049
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.1
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.5105,
          "n": 143,
          "acc": 0.4755,
          "acc_norm": 0.5035,
          "soft_score": 0.4825,
          "soft_score_norm": 0.5105,
          "stderr": 0.0417,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 0.8,
              "acc_norm": 0.8
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.6
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.5
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 0.8571,
              "acc_norm": 0.8571
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.8
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.8
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.2,
              "acc_norm": 0.1
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.2,
              "acc_norm": 0.4
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.4
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.3
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 0.875,
              "acc_norm": 0.875
            },
            "Math-pattern": {
              "n": 7,
              "acc": 0.7143,
              "acc_norm": 0.7143
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.2,
              "acc_norm": 0.1
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.5
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 0.75,
              "acc_norm": 0.375
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.5556,
              "acc_norm": 0.5556
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "liquidai-lfm2.5-350m-base",
      "name": "LiquidAI/LFM2.5-350M-Base",
      "org": "LiquidAI",
      "params_b": null,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/LiquidAI/LFM2.5-350M-Base",
      "model_revision": "9960764e30892e01f29a6dc23df2533fcd8bd5ae",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.0,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.0
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.0
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.0
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.0
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.2788,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.7048
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.6998
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.744
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.7958
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.7201
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.146
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.5558
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.0271
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.0
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.5105,
          "n": 143,
          "acc": 0.3986,
          "acc_norm": 0.5035,
          "soft_score": 0.4091,
          "soft_score_norm": 0.5105,
          "stderr": 0.0417,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 0.8,
              "acc_norm": 0.6
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.5
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.3
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 0.4286,
              "acc_norm": 0.5714
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.1,
              "acc_norm": 0.7
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.8
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.0,
              "acc_norm": 0.2
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.7
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.6
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.5
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 0.875,
              "acc_norm": 0.875
            },
            "Math-pattern": {
              "n": 7,
              "acc": 0.7143,
              "acc_norm": 0.7143
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.3
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.5
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 0.625,
              "acc_norm": 0.375
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.2222,
              "acc_norm": 0.2222
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "liquidai-lfm2.5-230m",
      "name": "LiquidAI/LFM2.5-230M",
      "org": "LiquidAI",
      "params_b": 0.23,
      "license": null,
      "architecture": "lfm2.5",
      "url": "https://huggingface.co/LiquidAI/LFM2.5-230M",
      "model_revision": "37b30cce3446f3f2e26a0d3f8c67c9167f5079d7",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.0,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.0
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.0
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.0
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.0
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.2953,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.7178
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.738
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.7269
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.7906
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.7643
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.302
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.7019
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.049
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.0
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.514,
          "n": 143,
          "acc": 0.4825,
          "acc_norm": 0.5105,
          "soft_score": 0.486,
          "soft_score_norm": 0.514,
          "stderr": 0.0417,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 0.6,
              "acc_norm": 0.6
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.5
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.4
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 0.8571,
              "acc_norm": 0.8571
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.7
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.7
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.2,
              "acc_norm": 0.1
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.4
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.2,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.6
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.4
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 0.875,
              "acc_norm": 0.875
            },
            "Math-pattern": {
              "n": 7,
              "acc": 0.5714,
              "acc_norm": 0.5714
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.3
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.5
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 0.625,
              "acc_norm": 0.75
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.4444,
              "acc_norm": 0.4444
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "qwen-qwen2.5-1.5b",
      "name": "Qwen/Qwen2.5-1.5B",
      "org": "Qwen",
      "params_b": 1.5,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/Qwen/Qwen2.5-1.5B",
      "model_revision": "8faed761d45a263340a0528343f099c05c9a4323",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.0,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.0
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.0
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.0
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.0
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.2563,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.5943
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.7341
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.7074
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.676
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.6957
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.2408
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.3879
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.1135
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.0
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.7517,
          "n": 143,
          "acc": 0.7483,
          "acc_norm": 0.7483,
          "soft_score": 0.7545,
          "soft_score_norm": 0.7517,
          "stderr": 0.036,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 1.0,
              "acc_norm": 0.8
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.8
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.7,
              "acc_norm": 0.9
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.6
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.9,
              "acc_norm": 0.9
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.2
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.4
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.3
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 0.875,
              "acc_norm": 0.875
            },
            "Math-pattern": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.8,
              "acc_norm": 0.6
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.5556,
              "acc_norm": 0.5556
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "qwen-qwen2.5-0.5b-instruct",
      "name": "Qwen/Qwen2.5-0.5B-Instruct",
      "org": "Qwen",
      "params_b": 0.5,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/Qwen/Qwen2.5-0.5B-Instruct",
      "model_revision": "7ae557604adf67be50417f59c2c2f167def9a775",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.225,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.1905
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.119
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0238
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.825
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.1892
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.4847,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.7134
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.7698
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.7428
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.2424
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.8114
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.7592
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.37
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.9131
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.1
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.8485
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.1892
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.2
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.6636,
          "n": 143,
          "acc": 0.6154,
          "acc_norm": 0.6573,
          "soft_score": 0.6252,
          "soft_score_norm": 0.6636,
          "stderr": 0.0394,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.9,
              "acc_norm": 0.7
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.7
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.7,
              "acc_norm": 0.9
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.8
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.2,
              "acc_norm": 0.3
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.6,
              "acc_norm": 0.8
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.2
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.4
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.4
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Math-pattern": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.2
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.75
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 0.875
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.4444,
              "acc_norm": 0.4444
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "qwen-qwen2.5-0.5b",
      "name": "Qwen/Qwen2.5-0.5B",
      "org": "Qwen",
      "params_b": 0.5,
      "license": null,
      "architecture": null,
      "url": "https://huggingface.co/Qwen/Qwen2.5-0.5B",
      "model_revision": "060db6499f32faf8b98477b0a26969ef7d8b9987",
      "script_sha256": "955411f9971c4c26a6eeb3aa43b1694fe66ac430a139011faa915543746d2921",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.075,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.0
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.0238
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.0
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.0
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.425
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.0
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.314,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.6992
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.6406
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.6482
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.0
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.7609
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.6587
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.3333
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.4212
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.0
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.3333
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.0
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.0538
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.0
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.5986,
          "n": 143,
          "acc": 0.5594,
          "acc_norm": 0.5874,
          "soft_score": 0.5657,
          "soft_score_norm": 0.5986,
          "stderr": 0.0406,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 0.9,
              "acc_norm": 0.6
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.7
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.7,
              "acc_norm": 0.9
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.7,
              "acc_norm": 0.9
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.1,
              "acc_norm": 0.1
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.7
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.2
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.4
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.1,
              "acc_norm": 0.2
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 0.875,
              "acc_norm": 0.875
            },
            "Math-pattern": {
              "n": 7,
              "acc": 0.7143,
              "acc_norm": 0.7143
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.4,
              "acc_norm": 0.2
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.75
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 0.625
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.4444,
              "acc_norm": 0.4444
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        },
        "bench-effortless-7-2026": {
          "score": 0.0033,
          "n": 300,
          "stderr": 0.0033,
          "metrics": {
            "generative": {
              "exact_match": 0.0033,
              "hybrid_score": 0.2351
            },
            "loglikelihood": {
              "acc": 0.7733,
              "acc_norm": 0.7433,
              "soft_score": 0.7733,
              "soft_score_norm": 0.7433
            }
          },
          "categories": {
            "Commonsense-causality": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7778
            },
            "Commonsense-reasoning": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7778
            },
            "Commonsense-simulation": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7222
            },
            "Knowledge-basic": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7778
            },
            "Knowledge-definitions": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.8889
            },
            "Language-comprehension": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.8333
            },
            "Language-structure": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7222
            },
            "Language-transformation": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7778
            },
            "Logic-consistency": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.9444
            },
            "Logic-pattern": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7778
            },
            "Math-arithmetic": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.8824
            },
            "Math-pattern": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.9412
            },
            "Math-reasoning": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7059
            },
            "Pattern-generation": {
              "n": 17,
              "score": 0.0588,
              "exact_match": 0.0588,
              "acc_norm": 0.8235
            },
            "Pattern-matching": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7059
            },
            "Pattern-recognition": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.5882
            }
          }
        },
        "bench-easy-7-2026": {
          "score": 0.266,
          "n": 300,
          "stderr": 0.02,
          "metrics": {
            "generative": {
              "exact_match": 0.02,
              "hybrid_score": 0.266
            },
            "loglikelihood": {
              "acc": 0.6767,
              "acc_norm": 0.64,
              "soft_score": 0.6767,
              "soft_score_norm": 0.64
            }
          },
          "categories": {
            "Commonsense-causality": {
              "n": 18,
              "score": 0.6793,
              "exact_match": 0.0,
              "acc_norm": 0.8333
            },
            "Commonsense-reasoning": {
              "n": 18,
              "score": 0.6656,
              "exact_match": 0.0,
              "acc_norm": 0.7778
            },
            "Commonsense-simulation": {
              "n": 18,
              "score": 0.6951,
              "exact_match": 0.0,
              "acc_norm": 0.5556
            },
            "Knowledge-basic": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7778
            },
            "Knowledge-definitions": {
              "n": 18,
              "score": 0.7592,
              "exact_match": 0.0,
              "acc_norm": 1.0
            },
            "Language-comprehension": {
              "n": 18,
              "score": 0.7092,
              "exact_match": 0.0,
              "acc_norm": 0.8889
            },
            "Language-structure": {
              "n": 18,
              "score": 0.131,
              "exact_match": 0.0,
              "acc_norm": 0.5556
            },
            "Language-transformation": {
              "n": 18,
              "score": 0.5156,
              "exact_match": 0.0556,
              "acc_norm": 0.7778
            },
            "Logic-consistency": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 18,
              "score": 0.0556,
              "exact_match": 0.0556,
              "acc_norm": 0.7778
            },
            "Logic-pattern": {
              "n": 18,
              "score": 0.1111,
              "exact_match": 0.1111,
              "acc_norm": 0.6667
            },
            "Math-arithmetic": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1765
            },
            "Math-pattern": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7059
            },
            "Math-reasoning": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.7647
            },
            "Pattern-generation": {
              "n": 17,
              "score": 0.1176,
              "exact_match": 0.1176,
              "acc_norm": 0.4706
            },
            "Pattern-matching": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.5882
            },
            "Pattern-recognition": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.5294
            }
          }
        },
        "bench-mid-7-2026": {
          "score": 0.5033,
          "n": 300,
          "stderr": 0.0289,
          "metrics": {
            "generative": {
              "exact_match": 0.0033,
              "hybrid_score": 0.2509
            },
            "loglikelihood": {
              "acc": 0.49,
              "acc_norm": 0.5033,
              "soft_score": 0.49,
              "soft_score_norm": 0.5033
            }
          },
          "categories": {
            "Commonsense-causality": {
              "n": 18,
              "score": 0.4444,
              "exact_match": 0.0,
              "acc_norm": 0.4444
            },
            "Commonsense-reasoning": {
              "n": 18,
              "score": 0.5556,
              "exact_match": 0.0,
              "acc_norm": 0.5556
            },
            "Commonsense-simulation": {
              "n": 18,
              "score": 0.5556,
              "exact_match": 0.0,
              "acc_norm": 0.5556
            },
            "Knowledge-basic": {
              "n": 18,
              "score": 0.8333,
              "exact_match": 0.0,
              "acc_norm": 0.8333
            },
            "Knowledge-definitions": {
              "n": 18,
              "score": 0.8333,
              "exact_match": 0.0,
              "acc_norm": 0.8333
            },
            "Language-comprehension": {
              "n": 18,
              "score": 0.6667,
              "exact_match": 0.0,
              "acc_norm": 0.6667
            },
            "Language-structure": {
              "n": 18,
              "score": 0.6667,
              "exact_match": 0.0,
              "acc_norm": 0.6667
            },
            "Language-transformation": {
              "n": 18,
              "score": 0.7222,
              "exact_match": 0.0,
              "acc_norm": 0.7222
            },
            "Logic-consistency": {
              "n": 18,
              "score": 0.1111,
              "exact_match": 0.0,
              "acc_norm": 0.1111
            },
            "Logic-deduction": {
              "n": 18,
              "score": 0.6111,
              "exact_match": 0.0,
              "acc_norm": 0.6111
            },
            "Logic-pattern": {
              "n": 18,
              "score": 0.4444,
              "exact_match": 0.0,
              "acc_norm": 0.4444
            },
            "Math-arithmetic": {
              "n": 17,
              "score": 0.2353,
              "exact_match": 0.0,
              "acc_norm": 0.2353
            },
            "Math-pattern": {
              "n": 17,
              "score": 0.5882,
              "exact_match": 0.0588,
              "acc_norm": 0.5882
            },
            "Math-reasoning": {
              "n": 17,
              "score": 0.1176,
              "exact_match": 0.0,
              "acc_norm": 0.1176
            },
            "Pattern-generation": {
              "n": 17,
              "score": 0.4118,
              "exact_match": 0.0,
              "acc_norm": 0.4118
            },
            "Pattern-matching": {
              "n": 17,
              "score": 0.4706,
              "exact_match": 0.0,
              "acc_norm": 0.4706
            },
            "Pattern-recognition": {
              "n": 17,
              "score": 0.2353,
              "exact_match": 0.0,
              "acc_norm": 0.2353
            }
          }
        }
      }
    },
    {
      "id": "qwen-qwen2.5-1.5b-instruct",
      "name": "Qwen/Qwen2.5-1.5B-Instruct",
      "org": "Qwen",
      "params_b": 1.5,
      "license": "Apache 2.0",
      "architecture": null,
      "url": "https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct",
      "model_revision": "989aa7980e4cf806f80c7fef2b1adb7bc71aa306",
      "script_sha256": "0c557dde5397ded8568280d621300759b4387a9f12227817c619a9325be8806a",
      "runs": {
        "bench-effortless-6-2026": {
          "score": 0.7458,
          "n": 240,
          "notes": "Exact-match, normalized.",
          "categories": {
            "Commonsense-reasoning": {
              "n": 37,
              "exact_match": 0.5405
            },
            "Knowledge-basic": {
              "n": 42,
              "exact_match": 0.5714
            },
            "Language-comprehension": {
              "n": 42,
              "exact_match": 0.6667
            },
            "Logic-deduction": {
              "n": 42,
              "exact_match": 0.881
            },
            "Math-arithmetic": {
              "n": 40,
              "exact_match": 0.925
            },
            "Pattern-recognition": {
              "n": 37,
              "exact_match": 0.8919
            }
          }
        },
        "bench-easy-6-2026": {
          "score": 0.7463,
          "n": 238,
          "notes": "Hybrid category-aware scoring (strict / flexible / semantic).",
          "categories": {
            "Commonsense-causality": {
              "n": 10,
              "hybrid_score": 0.8323
            },
            "Commonsense-reasoning": {
              "n": 10,
              "hybrid_score": 0.8156
            },
            "Commonsense-simulation": {
              "n": 10,
              "hybrid_score": 0.7773
            },
            "Knowledge-basic": {
              "n": 33,
              "hybrid_score": 0.7879
            },
            "Knowledge-definitions": {
              "n": 38,
              "hybrid_score": 0.8114
            },
            "Language-comprehension": {
              "n": 10,
              "hybrid_score": 0.7737
            },
            "Language-structure": {
              "n": 10,
              "hybrid_score": 0.7059
            },
            "Language-transformation": {
              "n": 10,
              "hybrid_score": 0.8567
            },
            "Logic-consistency": {
              "n": 10,
              "hybrid_score": 0.0
            },
            "Logic-deduction": {
              "n": 15,
              "hybrid_score": 0.4667
            },
            "Logic-pattern": {
              "n": 10,
              "hybrid_score": 0.5
            },
            "Math-arithmetic": {
              "n": 33,
              "hybrid_score": 0.9697
            },
            "Math-pattern": {
              "n": 14,
              "hybrid_score": 0.8571
            },
            "Math-reasoning": {
              "n": 15,
              "hybrid_score": 0.8786
            },
            "Pattern-matching": {
              "n": 10,
              "hybrid_score": 0.4
            }
          }
        },
        "bench-mid-6-2026": {
          "score": 0.7238,
          "n": 143,
          "acc": 0.6853,
          "acc_norm": 0.7203,
          "soft_score": 0.6951,
          "soft_score_norm": 0.7238,
          "stderr": 0.0373,
          "categories": {
            "Commonsense-causality": {
              "n": 5,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Commonsense-reasoning": {
              "n": 10,
              "acc": 1.0,
              "acc_norm": 0.8
            },
            "Commonsense-simulation": {
              "n": 10,
              "acc": 0.7,
              "acc_norm": 0.9
            },
            "Knowledge-basic": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Knowledge-definitions": {
              "n": 10,
              "acc": 0.8,
              "acc_norm": 1.0
            },
            "Language-comprehension": {
              "n": 10,
              "acc": 0.7,
              "acc_norm": 0.7
            },
            "Language-structure": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.6
            },
            "Language-transformation": {
              "n": 10,
              "acc": 0.5,
              "acc_norm": 0.6
            },
            "Logic-consistency": {
              "n": 5,
              "acc": 0.0,
              "acc_norm": 0.2
            },
            "Logic-deduction": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.5
            },
            "Logic-pattern": {
              "n": 10,
              "acc": 0.3,
              "acc_norm": 0.2
            },
            "Math-arithmetic": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Math-pattern": {
              "n": 7,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Math-reasoning": {
              "n": 10,
              "acc": 0.7,
              "acc_norm": 0.6
            },
            "Pattern-generation": {
              "n": 4,
              "acc": 0.75,
              "acc_norm": 0.75
            },
            "Pattern-matching": {
              "n": 8,
              "acc": 1.0,
              "acc_norm": 1.0
            },
            "Pattern-recognition": {
              "n": 9,
              "acc": 0.5556,
              "acc_norm": 0.5556
            }
          }
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    },
    {
      "id": "atomixlabs-photon-1.0-1m",
      "name": "AtomixLabs/Photon-1.0-1M",
      "org": "AtomixLabs",
      "params_b": 0.001,
      "license": "apache-2.0",
      "architecture": "LlamaForCausalLM",
      "url": "https://huggingface.co/AtomixLabs/Photon-1.0-1M",
      "model_revision": "a03eab42a7debdc27b797f8f5904a689720d0de4",
      "script_sha256": "955411f9971c4c26a6eeb3aa43b1694fe66ac430a139011faa915543746d2921",
      "runs": {
        "bench-effortless-7-2026": {
          "score": 0.0,
          "n": 300,
          "stderr": 0.0,
          "metrics": {
            "generative": {
              "exact_match": 0.0,
              "hybrid_score": 0.1942
            },
            "loglikelihood": {
              "acc": 0.33,
              "acc_norm": 0.2733,
              "soft_score": 0.33,
              "soft_score_norm": 0.2733
            }
          },
          "categories": {
            "Commonsense-causality": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1111
            },
            "Commonsense-reasoning": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.4444
            },
            "Commonsense-simulation": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Knowledge-basic": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1111
            },
            "Knowledge-definitions": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.3889
            },
            "Language-comprehension": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.5556
            },
            "Language-structure": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1667
            },
            "Language-transformation": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.3889
            },
            "Logic-consistency": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Logic-pattern": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.0556
            },
            "Math-arithmetic": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.3529
            },
            "Math-pattern": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1176
            },
            "Math-reasoning": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1765
            },
            "Pattern-generation": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.6471
            },
            "Pattern-matching": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2941
            },
            "Pattern-recognition": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2941
            }
          }
        },
        "bench-easy-7-2026": {
          "score": 0.203,
          "n": 300,
          "stderr": 0.0158,
          "metrics": {
            "generative": {
              "exact_match": 0.0,
              "hybrid_score": 0.203
            },
            "loglikelihood": {
              "acc": 0.2833,
              "acc_norm": 0.31,
              "soft_score": 0.2833,
              "soft_score_norm": 0.31
            }
          },
          "categories": {
            "Commonsense-causality": {
              "n": 18,
              "score": 0.5875,
              "exact_match": 0.0,
              "acc_norm": 0.2222
            },
            "Commonsense-reasoning": {
              "n": 18,
              "score": 0.5981,
              "exact_match": 0.0,
              "acc_norm": 0.6111
            },
            "Commonsense-simulation": {
              "n": 18,
              "score": 0.6044,
              "exact_match": 0.0,
              "acc_norm": 0.1667
            },
            "Knowledge-basic": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2222
            },
            "Knowledge-definitions": {
              "n": 18,
              "score": 0.6303,
              "exact_match": 0.0,
              "acc_norm": 0.3889
            },
            "Language-comprehension": {
              "n": 18,
              "score": 0.5766,
              "exact_match": 0.0,
              "acc_norm": 0.5556
            },
            "Language-structure": {
              "n": 18,
              "score": 0.0829,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Language-transformation": {
              "n": 18,
              "score": 0.303,
              "exact_match": 0.0,
              "acc_norm": 0.5
            },
            "Logic-consistency": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.3333
            },
            "Logic-pattern": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Math-arithmetic": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.3529
            },
            "Math-pattern": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2353
            },
            "Math-reasoning": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1765
            },
            "Pattern-generation": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.5294
            },
            "Pattern-matching": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.1765
            },
            "Pattern-recognition": {
              "n": 17,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.2353
            }
          }
        },
        "bench-mid-7-2026": {
          "score": 0.2867,
          "n": 300,
          "stderr": 0.0262,
          "metrics": {
            "generative": {
              "exact_match": 0.0,
              "hybrid_score": 0.205
            },
            "loglikelihood": {
              "acc": 0.1933,
              "acc_norm": 0.2867,
              "soft_score": 0.1933,
              "soft_score_norm": 0.2867
            }
          },
          "categories": {
            "Commonsense-causality": {
              "n": 18,
              "score": 0.2778,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Commonsense-reasoning": {
              "n": 18,
              "score": 0.4444,
              "exact_match": 0.0,
              "acc_norm": 0.4444
            },
            "Commonsense-simulation": {
              "n": 18,
              "score": 0.2778,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Knowledge-basic": {
              "n": 18,
              "score": 0.2778,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Knowledge-definitions": {
              "n": 18,
              "score": 0.4444,
              "exact_match": 0.0,
              "acc_norm": 0.4444
            },
            "Language-comprehension": {
              "n": 18,
              "score": 0.3889,
              "exact_match": 0.0,
              "acc_norm": 0.3889
            },
            "Language-structure": {
              "n": 18,
              "score": 0.2778,
              "exact_match": 0.0,
              "acc_norm": 0.2778
            },
            "Language-transformation": {
              "n": 18,
              "score": 0.1667,
              "exact_match": 0.0,
              "acc_norm": 0.1667
            },
            "Logic-consistency": {
              "n": 18,
              "score": 0.0,
              "exact_match": 0.0,
              "acc_norm": 0.0
            },
            "Logic-deduction": {
              "n": 18,
              "score": 0.5,
              "exact_match": 0.0,
              "acc_norm": 0.5
            },
            "Logic-pattern": {
              "n": 18,
              "score": 0.2222,
              "exact_match": 0.0,
              "acc_norm": 0.2222
            },
            "Math-arithmetic": {
              "n": 17,
              "score": 0.0588,
              "exact_match": 0.0,
              "acc_norm": 0.0588
            },
            "Math-pattern": {
              "n": 17,
              "score": 0.1176,
              "exact_match": 0.0,
              "acc_norm": 0.1176
            },
            "Math-reasoning": {
              "n": 17,
              "score": 0.0588,
              "exact_match": 0.0,
              "acc_norm": 0.0588
            },
            "Pattern-generation": {
              "n": 17,
              "score": 0.5882,
              "exact_match": 0.0,
              "acc_norm": 0.5882
            },
            "Pattern-matching": {
              "n": 17,
              "score": 0.4118,
              "exact_match": 0.0,
              "acc_norm": 0.4118
            },
            "Pattern-recognition": {
              "n": 17,
              "score": 0.3529,
              "exact_match": 0.0,
              "acc_norm": 0.3529
            }
          }
        },
        "bench-effortless-6-2026": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        },
        "bench-easy-6-2026": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        },
        "bench-mid-6-2026": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        },
        "bench-AGI": {
          "score": null,
          "n": null,
          "notes": "Not yet evaluated on this tier."
        }
      }
    }
  ]
}