{
  "vocabulary_version": 1,
  "contract_version": "2.1",
  "snapshot": "snap_3332adf23fd8f9cf",
  "default_task_tokens": {
    "input": 40000,
    "output": 4000
  },
  "task_types": [
    "new_feature",
    "bug_fix",
    "refactor",
    "test_writing",
    "docs",
    "migration",
    "performance",
    "security_fix",
    "review",
    "analysis",
    "data_transform",
    "config_infra"
  ],
  "facets": [
    {
      "id": "model.class",
      "label": "Model class",
      "definition": "The model's class as derived from its model type by `api/classes.py`: what it consumes, what it emits and what decision it makes. One value per model. Class is derived, not authored, and says nothing about quality.",
      "subject": "model",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 44,
      "of": 44,
      "values": [
        {
          "value": "decider",
          "count": 1,
          "label": "Decision model"
        },
        {
          "value": "orderer",
          "count": 2,
          "label": "Reranker"
        },
        {
          "value": "text-generator",
          "count": 35,
          "label": "Text generator"
        },
        {
          "value": "vectoriser",
          "count": 6,
          "label": "Embedding model"
        }
      ]
    },
    {
      "id": "model.input_modalities",
      "label": "Input modalities",
      "definition": "Every modality the model accepts as input natively, without a separate model transcribing or captioning it first. `document` means files such as PDF read as documents, not text extracted by the caller.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 33,
      "of": 44,
      "values": [
        {
          "value": "audio",
          "count": 7,
          "label": "Audio"
        },
        {
          "value": "document",
          "count": 17,
          "label": "Documents"
        },
        {
          "value": "image",
          "count": 23,
          "label": "Images"
        },
        {
          "value": "text",
          "count": 33,
          "label": "Text"
        },
        {
          "value": "video",
          "count": 9,
          "label": "Video"
        }
      ]
    },
    {
      "id": "model.output_modalities",
      "label": "Output modalities",
      "definition": "Every modality the model emits natively. `embedding` is a vector, `score` a relevance or reward number, `label` a class from a fixed or caller-defined set, and `action` a tool or environment action.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 33,
      "of": 44,
      "values": [
        {
          "value": "embedding",
          "count": 4,
          "label": "Embeddings"
        },
        {
          "value": "label",
          "count": 1,
          "label": "Labels"
        },
        {
          "value": "score",
          "count": 3,
          "label": "Scores"
        },
        {
          "value": "text",
          "count": 26,
          "label": "Text"
        }
      ]
    },
    {
      "id": "model.context_window",
      "label": "Context window",
      "definition": "The maximum number of tokens the model accepts in one request, input and output together, as documented by its lab for its largest supported configuration. A provider that serves a smaller window records that on the offering, not here.",
      "subject": "model",
      "value_type": "number",
      "unit": "tokens",
      "unit_definition": "Tokens as counted by the model's own tokenizer, as documented by its lab. Not words, characters or another model's tokens.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 34,
      "of": 44,
      "range": {
        "min": 8192,
        "max": 1050000
      }
    },
    {
      "id": "model.max_output_tokens",
      "label": "Maximum output tokens",
      "definition": "The maximum number of tokens the model can emit in one response, including any reasoning tokens the lab counts against the limit, as documented by its lab.",
      "subject": "model",
      "value_type": "number",
      "unit": "tokens",
      "unit_definition": "Tokens as counted by the model's own tokenizer, as documented by its lab. Not words, characters or another model's tokens.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 21,
      "of": 44,
      "range": {
        "min": 65536,
        "max": 262144
      }
    },
    {
      "id": "model.weights_openness",
      "label": "Open weights",
      "definition": "`open_weights` when the lab publishes the trained weights for download under any licence, including a gated or restrictive one; `closed_weights` otherwise. Says nothing about the licence terms, which are the `licence.*` facets.",
      "subject": "model",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 40,
      "of": 44,
      "values": [
        {
          "value": "closed_weights",
          "count": 21,
          "label": "Closed weights"
        },
        {
          "value": "open_weights",
          "count": 19,
          "label": "Open weights"
        }
      ]
    },
    {
      "id": "model.parameters_total",
      "label": "Total parameters",
      "definition": "The total number of parameters in the released model, counting every expert of a mixture-of-experts model. `not_disclosed` is a state of the fact, never an estimate.",
      "subject": "model",
      "value_type": "number",
      "unit": "parameters",
      "unit_definition": "A count of trainable weights (not billions; 7e9 is written 7000000000).",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 19,
      "of": 44,
      "range": {
        "min": 4021782018,
        "max": 2779931837184
      }
    },
    {
      "id": "model.parameters_active",
      "label": "Active parameters",
      "definition": "The number of parameters used to process one token. Equal to the total for a dense model; for a mixture-of-experts model, as the lab documents it.",
      "subject": "model",
      "value_type": "number",
      "unit": "parameters",
      "unit_definition": "A count of trainable weights (not billions; 7e9 is written 7000000000).",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 44,
      "range": null
    },
    {
      "id": "model.architecture",
      "label": "Architecture",
      "definition": "The model's architecture family as `ArchitectureType` in `schema/enums.py` names it. An undisclosed architecture is unknown, never inapplicable.",
      "subject": "model",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 44,
      "values": []
    },
    {
      "id": "licence.commercial_use",
      "label": "Commercial use",
      "definition": "Whether the licence governing the model lets a customer use the model or its outputs in a commercial product. `permitted_with_conditions` covers any condition (user caps, attribution, field-of-use limits); the other `licence.*` facets say which.",
      "subject": "model",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 29,
      "of": 44,
      "values": [
        {
          "value": "permitted",
          "count": 6,
          "label": "Permitted"
        },
        {
          "value": "permitted_with_conditions",
          "count": 23,
          "label": "Permitted with conditions"
        }
      ]
    },
    {
      "id": "licence.user_cap",
      "label": "Licence user cap",
      "definition": "The largest number of monthly active users a licensee may serve before the licence requires a separate agreement. `unbounded` when the licence sets no such cap.",
      "subject": "model",
      "value_type": "number",
      "unit": "monthly_active_users",
      "unit_definition": "Distinct end users in a calendar month, as the licence itself defines the count.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "governance",
      "computed_by": null,
      "known": 27,
      "of": 44,
      "range": null,
      "literals": [
        "unbounded"
      ]
    },
    {
      "id": "licence.output_training",
      "label": "Training on outputs",
      "definition": "Whether the licence or terms let a customer use the model's outputs to train or improve another model. `restricted` when permitted only for some purposes or models (for example not for a competing model).",
      "subject": "model",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 28,
      "of": 44,
      "values": [
        {
          "value": "permitted",
          "count": 11,
          "label": "Permitted"
        },
        {
          "value": "restricted",
          "count": 17,
          "label": "Restricted"
        }
      ]
    },
    {
      "id": "licence.fine_tuning",
      "label": "Fine-tuning rights",
      "definition": "Whether the licence lets a customer modify the model's weights by further training and use the result. Whether a provider offers fine-tuning as a service is `offering.fine_tuning`, a different fact.",
      "subject": "model",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 27,
      "of": 44,
      "values": [
        {
          "value": "permitted",
          "count": 6,
          "label": "Permitted"
        },
        {
          "value": "permitted_with_conditions",
          "count": 5,
          "label": "Permitted with conditions"
        },
        {
          "value": "prohibited",
          "count": 16,
          "label": "Prohibited"
        }
      ]
    },
    {
      "id": "origin.lab_jurisdiction",
      "label": "Lab jurisdiction",
      "definition": "The country or countries where the lab that trained this model is incorporated, as ISO 3166-1 alpha-2 codes. Where the lab has a parent company, the parent's country is included too. Says nothing about where the model was based on or where it runs.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "governance",
      "computed_by": null,
      "known": 21,
      "of": 44,
      "values": [
        {
          "value": "US",
          "count": 21
        }
      ]
    },
    {
      "id": "origin.base_lineage",
      "label": "Base model lineage",
      "definition": "The countries of incorporation of the labs that trained every model this model's weights were derived from (fine-tuned, distilled into, merged or continued from), as ISO 3166-1 alpha-2 codes. The empty set when the model was trained from random initialisation. Excludes this model's own lab unless it also trained an ancestor.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "governance",
      "computed_by": null,
      "known": 5,
      "of": 44,
      "values": [
        {
          "value": "CN",
          "count": 4
        },
        {
          "value": "US",
          "count": 1
        }
      ]
    },
    {
      "id": "origin.weights_hosting",
      "label": "Weights hosted in",
      "definition": "The countries in which the lab itself stores and serves the weights for the hosted inference it sells first-party, as ISO 3166-1 alpha-2 codes. The empty set when the lab sells no hosted inference of this model. Where a third-party provider runs the model is `offering.region`, not this.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "governance",
      "computed_by": null,
      "known": 0,
      "of": 44,
      "values": []
    },
    {
      "id": "origin.base_models",
      "label": "Base models",
      "definition": "The `lab/model-id` of every model this model's weights were directly derived from. The empty set when trained from random initialisation. Backs `origin.base_lineage` with the models themselves.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "governance",
      "computed_by": null,
      "known": 0,
      "of": 44,
      "values": []
    },
    {
      "id": "model.release_date",
      "label": "Release date",
      "definition": "The date the lab first made the model generally available to the public, by API or download. A preview or waitlist release counts only if anyone could sign up.",
      "subject": "model",
      "value_type": "date",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 32,
      "of": 44,
      "range": {
        "min": "2025-08-01",
        "max": "2026-09-28"
      }
    },
    {
      "id": "model.lifecycle",
      "label": "Lifecycle",
      "definition": "`deprecated` when the lab has announced a retirement date; `retired` when the lab no longer serves or supports the model; `active` otherwise. Retired models are excluded from decisions unless a spec asks for them.",
      "subject": "model",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 44,
      "of": 44,
      "values": [
        {
          "value": "active",
          "count": 44,
          "label": "Active"
        }
      ]
    },
    {
      "id": "model.knowledge_cutoff",
      "label": "Knowledge cutoff",
      "definition": "The latest date of training data the lab documents for the model. A month is recorded as its first day, and the fact's qualifiers say so.",
      "subject": "model",
      "value_type": "date",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 44,
      "range": null
    },
    {
      "id": "model.deprecation_date",
      "label": "Deprecation date",
      "definition": "The date the lab has announced the model will stop being served by its own API. Unknown until announced; not the date of the announcement.",
      "subject": "model",
      "value_type": "date",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 44,
      "range": null
    },
    {
      "id": "feature.tool_calling",
      "label": "Tool calling",
      "definition": "True when the lab documents that the model emits structured function or tool calls against caller-supplied tool definitions through at least one first-party interface.",
      "subject": "model",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 34,
      "of": 44,
      "values": [
        {
          "value": true,
          "count": 34
        }
      ]
    },
    {
      "id": "feature.structured_output",
      "label": "Structured output",
      "definition": "True when the lab documents constrained output that conforms to a caller-supplied JSON schema. A prompt-only \"JSON mode\" without a schema guarantee is false.",
      "subject": "model",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 27,
      "of": 44,
      "values": [
        {
          "value": true,
          "count": 27
        }
      ]
    },
    {
      "id": "feature.effort_controls",
      "label": "Effort controls",
      "definition": "True when the lab documents a request parameter that changes how much reasoning the model does (a reasoning effort level or thinking budget).",
      "subject": "model",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 33,
      "of": 44,
      "values": [
        {
          "value": true,
          "count": 33
        }
      ]
    },
    {
      "id": "feature.batch",
      "label": "Batch interface",
      "definition": "True when the lab's first-party API offers an asynchronous batch interface for this model. Batch pricing is `offering.price.batch_*`.",
      "subject": "model",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 19,
      "of": 44,
      "values": [
        {
          "value": true,
          "count": 19
        }
      ]
    },
    {
      "id": "feature.streaming",
      "label": "Streaming",
      "definition": "True when the lab's first-party API can return the response incrementally as it is generated.",
      "subject": "model",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 26,
      "of": 44,
      "values": [
        {
          "value": true,
          "count": 26
        }
      ]
    },
    {
      "id": "model.languages",
      "label": "Languages",
      "definition": "The natural languages the lab documents the model as supporting, as BCP 47 tags. Absence from the set means undocumented, not unsupported.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 3,
      "of": 44,
      "values": [
        {
          "value": "en",
          "count": 3
        }
      ]
    },
    {
      "id": "model.fits_hardware",
      "label": "Fits hardware",
      "definition": "The hardware SKUs in `hardware/` on which some published quantisation of the model's weights is estimated to load and run. An estimate, always labelled as one.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 40,
      "of": 44,
      "values": [
        {
          "value": "amd_instinct_mi210",
          "count": 11
        },
        {
          "value": "amd_instinct_mi250x",
          "count": 12
        },
        {
          "value": "amd_instinct_mi300x",
          "count": 12
        },
        {
          "value": "amd_instinct_mi325x",
          "count": 12
        },
        {
          "value": "amd_instinct_mi355x",
          "count": 12
        },
        {
          "value": "amd_rx_7900_xt",
          "count": 10
        },
        {
          "value": "amd_rx_7900_xtx",
          "count": 11
        },
        {
          "value": "amd_rx_9070_xt",
          "count": 8
        },
        {
          "value": "amd_ryzen_ai_max_plus_395",
          "count": 12
        },
        {
          "value": "apple_m1_max",
          "count": 11
        },
        {
          "value": "apple_m2_max",
          "count": 11
        },
        {
          "value": "apple_m2_ultra",
          "count": 12
        },
        {
          "value": "apple_m3_max",
          "count": 12
        },
        {
          "value": "apple_m3_ultra",
          "count": 16
        },
        {
          "value": "apple_m4",
          "count": 11
        },
        {
          "value": "apple_m4_max",
          "count": 12
        },
        {
          "value": "apple_m4_pro",
          "count": 11
        },
        {
          "value": "apple_m5_max",
          "count": 12
        },
        {
          "value": "google_tpu7x",
          "count": 12
        },
        {
          "value": "google_tpu_v4",
          "count": 11
        },
        {
          "value": "google_tpu_v5e",
          "count": 8
        },
        {
          "value": "google_tpu_v5p",
          "count": 11
        },
        {
          "value": "google_tpu_v6e",
          "count": 11
        },
        {
          "value": "intel_arc_a770_16gb",
          "count": 8
        },
        {
          "value": "intel_gaudi_2",
          "count": 11
        },
        {
          "value": "intel_gaudi_3",
          "count": 12
        },
        {
          "value": "nvidia_a100_40gb_sxm",
          "count": 11
        },
        {
          "value": "nvidia_a100_80gb_sxm",
          "count": 11
        },
        {
          "value": "nvidia_b200",
          "count": 12
        },
        {
          "value": "nvidia_b300",
          "count": 12
        },
        {
          "value": "nvidia_dgx_spark",
          "count": 12
        },
        {
          "value": "nvidia_gb200_superchip",
          "count": 12
        },
        {
          "value": "nvidia_h100_nvl",
          "count": 11
        },
        {
          "value": "nvidia_h100_pcie",
          "count": 11
        },
        {
          "value": "nvidia_h100_sxm",
          "count": 11
        },
        {
          "value": "nvidia_h200_sxm",
          "count": 12
        },
        {
          "value": "nvidia_jetson_agx_orin_32gb",
          "count": 11
        },
        {
          "value": "nvidia_jetson_agx_orin_64gb",
          "count": 11
        },
        {
          "value": "nvidia_jetson_orin_nano_4gb",
          "count": 3
        },
        {
          "value": "nvidia_jetson_orin_nano_8gb",
          "count": 7
        },
        {
          "value": "nvidia_jetson_orin_nx_16gb",
          "count": 8
        },
        {
          "value": "nvidia_jetson_orin_nx_8gb",
          "count": 7
        },
        {
          "value": "nvidia_jetson_t4000",
          "count": 11
        },
        {
          "value": "nvidia_jetson_t5000",
          "count": 12
        },
        {
          "value": "nvidia_l4",
          "count": 11
        },
        {
          "value": "nvidia_l40s",
          "count": 11
        },
        {
          "value": "nvidia_rtx_3060_12gb",
          "count": 8
        },
        {
          "value": "nvidia_rtx_3090",
          "count": 11
        },
        {
          "value": "nvidia_rtx_4000_sff_ada",
          "count": 10
        },
        {
          "value": "nvidia_rtx_4060_ti_16gb",
          "count": 8
        },
        {
          "value": "nvidia_rtx_4070_ti_super",
          "count": 8
        },
        {
          "value": "nvidia_rtx_4080_super",
          "count": 8
        },
        {
          "value": "nvidia_rtx_4090",
          "count": 11
        },
        {
          "value": "nvidia_rtx_4500_ada",
          "count": 11
        },
        {
          "value": "nvidia_rtx_5000_ada",
          "count": 11
        },
        {
          "value": "nvidia_rtx_5080",
          "count": 8
        },
        {
          "value": "nvidia_rtx_5090",
          "count": 11
        },
        {
          "value": "nvidia_rtx_6000_ada",
          "count": 11
        },
        {
          "value": "nvidia_rtx_a6000",
          "count": 11
        },
        {
          "value": "nvidia_rubin_gpu",
          "count": 12
        },
        {
          "value": "nvidia_tesla_p40",
          "count": 11
        },
        {
          "value": "qualcomm_snapdragon_x_elite",
          "count": 11
        }
      ]
    },
    {
      "id": "model.hardware_fit_indeterminate",
      "label": "Hardware fit indeterminate",
      "definition": "The hardware SKUs in `hardware/` for which ModelSpec's single-memory-pool weight-fit formula does not apply. Membership conditions on `model.fits_hardware` treat these device legs as unknown, not as false.",
      "subject": "model",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 19,
      "of": 44,
      "values": [
        {
          "value": "cerebras_wse3",
          "count": 19
        },
        {
          "value": "nvidia_vera_rubin_superchip",
          "count": 19
        }
      ]
    },
    {
      "id": "offering.provider",
      "label": "Provider",
      "definition": "The provider in `registry/providers.yaml` that sells the offering and bills the customer for it.",
      "subject": "offering",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 43,
      "of": 43,
      "values": [
        {
          "value": "alibaba-model-studio",
          "count": 1
        },
        {
          "value": "anthropic",
          "count": 7
        },
        {
          "value": "aws-bedrock",
          "count": 3
        },
        {
          "value": "azure-ai-foundry",
          "count": 4
        },
        {
          "value": "deepseek",
          "count": 2
        },
        {
          "value": "google-gemini-api",
          "count": 5
        },
        {
          "value": "google-vertex-ai",
          "count": 11
        },
        {
          "value": "meta-model-api",
          "count": 2
        },
        {
          "value": "openai",
          "count": 4
        },
        {
          "value": "typesafe",
          "count": 1
        },
        {
          "value": "xai",
          "count": 1
        },
        {
          "value": "zai",
          "count": 2
        }
      ]
    },
    {
      "id": "offering.region",
      "label": "Inference region",
      "definition": "The countries in which the provider documents that inference for this offering runs, as ISO 3166-1 alpha-2 codes. A global or routed offering lists every country it may run in. The provider's own region name is a qualifier on the fact.",
      "subject": "offering",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "governance",
      "computed_by": null,
      "known": 43,
      "of": 43,
      "values": [
        {
          "value": "global",
          "count": 35,
          "label": "Global"
        },
        {
          "value": "global-cross-region",
          "count": 3,
          "label": "Global, cross-region routing"
        },
        {
          "value": "global-short-context",
          "count": 4,
          "label": "Global, short context"
        },
        {
          "value": "singapore",
          "count": 1
        }
      ]
    },
    {
      "id": "offering.tier",
      "label": "Account tier",
      "definition": "The account tier the offering is sold under. A tier is its own offering only when a guaranteed fact differs from the standard tier: price, data handling or an attestation (design §4.1).",
      "subject": "offering",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 43,
      "of": 43,
      "values": [
        {
          "value": "standard",
          "count": 43,
          "label": "Standard"
        }
      ]
    },
    {
      "id": "offering.price.input",
      "label": "Input price",
      "definition": "The list price of uncached input tokens for a synchronous request. Where the price varies with prompt length, the lowest band, with the band as a qualifier.",
      "subject": "offering",
      "value_type": "number",
      "unit": "usd_per_1m_tokens",
      "unit_definition": "United States dollars per 1,000,000 tokens, at the provider's list price for the offering, before discounts, credits or committed-use pricing. Tokens are counted by the provider's own tokenizer for that model.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 43,
      "of": 43,
      "range": {
        "min": 0.042,
        "max": 10.0
      }
    },
    {
      "id": "offering.price.output",
      "label": "Output price",
      "definition": "The list price of output tokens, including reasoning tokens the provider bills as output, for a synchronous request, lowest band as for input.",
      "subject": "offering",
      "value_type": "number",
      "unit": "usd_per_1m_tokens",
      "unit_definition": "United States dollars per 1,000,000 tokens, at the provider's list price for the offering, before discounts, credits or committed-use pricing. Tokens are counted by the provider's own tokenizer for that model.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 43,
      "of": 43,
      "range": {
        "min": 0.0,
        "max": 50.0
      }
    },
    {
      "id": "offering.price.cached_input",
      "label": "Cached input price",
      "definition": "The list price of input tokens read from the provider's prompt cache. Cache-write surcharges are a qualifier. `not_offered` when the provider has no prompt cache for this offering, which is a known value, not an unknown one.",
      "subject": "offering",
      "value_type": "number",
      "unit": "usd_per_1m_tokens",
      "unit_definition": "United States dollars per 1,000,000 tokens, at the provider's list price for the offering, before discounts, credits or committed-use pricing. Tokens are counted by the provider's own tokenizer for that model.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 41,
      "of": 43,
      "range": {
        "min": 0.006,
        "max": 1.0
      }
    },
    {
      "id": "offering.price.batch_input",
      "label": "Batch input price",
      "definition": "The list price of input tokens submitted through the provider's asynchronous batch interface. `not_offered` when the provider has no batch interface for this offering.",
      "subject": "offering",
      "value_type": "number",
      "unit": "usd_per_1m_tokens",
      "unit_definition": "United States dollars per 1,000,000 tokens, at the provider's list price for the offering, before discounts, credits or committed-use pricing. Tokens are counted by the provider's own tokenizer for that model.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 30,
      "of": 43,
      "range": {
        "min": 0.15,
        "max": 5.0
      },
      "literals": [
        "not_offered"
      ]
    },
    {
      "id": "offering.price.batch_output",
      "label": "Batch output price",
      "definition": "The list price of output tokens returned through the provider's asynchronous batch interface. `not_offered` when the provider has no batch interface for this offering.",
      "subject": "offering",
      "value_type": "number",
      "unit": "usd_per_1m_tokens",
      "unit_definition": "United States dollars per 1,000,000 tokens, at the provider's list price for the offering, before discounts, credits or committed-use pricing. Tokens are counted by the provider's own tokenizer for that model.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 30,
      "of": 43,
      "range": {
        "min": 1.25,
        "max": 25.0
      },
      "literals": [
        "not_offered"
      ]
    },
    {
      "id": "offering.cost_per_task",
      "label": "Cost per task",
      "definition": "What one task costs on this offering at list prices, computed per decision from the spec's `task_tokens`: (offering.price.input × input tokens + offering.price.output × output tokens) / 1,000,000. Unknown when either price is unknown. Never authored on a card.",
      "subject": "offering",
      "value_type": "number",
      "unit": "usd_per_task",
      "unit_definition": "United States dollars for one task, at list prices: the offering's input price times the spec's input tokens per task, plus its output price times the spec's output tokens per task, divided by 1,000,000. Excludes cached, batch and committed-use pricing.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": "MODEL-153",
      "known": 43,
      "of": 43,
      "range": {
        "min": 0.00168,
        "max": 0.6
      }
    },
    {
      "id": "offering.speed.time_to_first_token",
      "label": "Time to first token",
      "definition": "The median time from sending a request to receiving the first output token. Always carries the method: prompt length, output length, effort, region of the client, sample size and date.",
      "subject": "offering",
      "value_type": "number",
      "unit": "milliseconds",
      "unit_definition": "Milliseconds of wall-clock time.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "range": null
    },
    {
      "id": "offering.speed.throughput",
      "label": "Output throughput",
      "definition": "The median output throughput of one request after its first token. Always carries the method, as for time to first token. Self-hosted throughput is a labelled estimate only.",
      "subject": "offering",
      "value_type": "number",
      "unit": "tokens_per_second",
      "unit_definition": "Output tokens generated per second of wall-clock time, measured from the first output token to the last, for a single request.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "range": null
    },
    {
      "id": "offering.rate_limit.requests",
      "label": "Request rate limit",
      "definition": "The documented default request rate limit for a new account at this tier, at the lowest paid usage level the provider documents.",
      "subject": "offering",
      "value_type": "number",
      "unit": "requests_per_minute",
      "unit_definition": "Requests accepted per 60-second window for one account at the tier the offering names, before any limit increase negotiated by contract.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "range": null
    },
    {
      "id": "offering.rate_limit.tokens",
      "label": "Token rate limit",
      "definition": "The documented default token rate limit for a new account at this tier, at the lowest paid usage level the provider documents.",
      "subject": "offering",
      "value_type": "number",
      "unit": "tokens_per_minute",
      "unit_definition": "Tokens (input plus output unless the provider states otherwise, in which case the fact's qualifiers say so) accepted per 60-second window for one account at the tier the offering names.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "range": null
    },
    {
      "id": "offering.sla_uptime",
      "label": "SLA uptime",
      "definition": "The monthly uptime the provider commits to in a published service level agreement for this offering, with service credits. A status-page history is not an SLA.",
      "subject": "offering",
      "value_type": "number",
      "unit": "percent",
      "unit_definition": "A percentage from 0 to 100.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "range": null
    },
    {
      "id": "offering.data.retention",
      "label": "Data retention",
      "definition": "The longest period the provider's terms say it keeps prompts and outputs for this offering by default, for any purpose including abuse monitoring. 0 means not stored after the response. `unbounded` when the terms set no limit.",
      "subject": "offering",
      "value_type": "number",
      "unit": "days",
      "unit_definition": "Calendar days.",
      "operators": [
        "=",
        "!=",
        "<",
        "<=",
        ">",
        ">=",
        "between",
        "known"
      ],
      "objective": true,
      "preference": {
        "kind": "continuous",
        "directions": [
          "max",
          "min"
        ],
        "threshold": "where"
      },
      "risk": "governance",
      "computed_by": null,
      "known": 26,
      "of": 43,
      "range": {
        "min": 0,
        "max": 55
      }
    },
    {
      "id": "offering.data.trains_on_customer_data",
      "label": "Trains on customer data",
      "definition": "True when the provider's default terms for this offering let it use customer prompts or outputs to train or improve models. An opt-out the customer must take still makes this true.",
      "subject": "offering",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 38,
      "of": 43,
      "values": [
        {
          "value": false,
          "count": 38
        }
      ]
    },
    {
      "id": "offering.data.zero_retention",
      "label": "Zero retention available",
      "definition": "True when the provider documents that a customer can have prompts and outputs for this offering not stored at all, including for abuse monitoring, whether by default or on request.",
      "subject": "offering",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 15,
      "of": 43,
      "values": [
        {
          "value": false,
          "count": 4
        },
        {
          "value": true,
          "count": 11
        }
      ]
    },
    {
      "id": "offering.attestation.soc2",
      "label": "SOC 2 report",
      "definition": "The SOC 2 report type the provider documents for the service this offering runs on, when that report's scope covers the service. `none` when the provider documents no such report. Documented, never \"compliant\".",
      "subject": "offering",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 8,
      "of": 43,
      "values": [
        {
          "value": "type_2",
          "count": 8,
          "label": "SOC 2 Type II"
        }
      ]
    },
    {
      "id": "offering.attestation.baa",
      "label": "HIPAA BAA",
      "definition": "True when the provider documents that it will sign a HIPAA business associate agreement that covers this offering. Says nothing about whether a customer's use is compliant.",
      "subject": "offering",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 22,
      "of": 43,
      "values": [
        {
          "value": true,
          "count": 22
        }
      ]
    },
    {
      "id": "offering.attestation.fedramp",
      "label": "FedRAMP level",
      "definition": "The FedRAMP authorization level of the cloud service offering this offering runs in, as the FedRAMP marketplace lists it for this model. `none` when not listed.",
      "subject": "offering",
      "value_type": "enum",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "enum value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "values": []
    },
    {
      "id": "offering.attestation.iso_27001",
      "label": "ISO/IEC 27001",
      "definition": "True when the provider documents an ISO/IEC 27001 certificate whose scope covers the service this offering runs on.",
      "subject": "offering",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "governance",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "values": []
    },
    {
      "id": "offering.fine_tuning",
      "label": "Fine-tuning service",
      "definition": "True when the provider sells fine-tuning of this model as a service and serves the tuned model under this offering's terms.",
      "subject": "offering",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "values": []
    },
    {
      "id": "offering.private_deployment",
      "label": "Private deployment",
      "definition": "True when the provider sells dedicated capacity for this model that is not shared with other customers (provisioned throughput, a dedicated endpoint or deployment in the customer's own cloud account).",
      "subject": "offering",
      "value_type": "boolean",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "=",
        "!=",
        "known"
      ],
      "objective": false,
      "preference": {
        "kind": "value",
        "term": {
          "prefer": "boolean value",
          "weight": "positive number"
        }
      },
      "risk": "capability",
      "computed_by": null,
      "known": 11,
      "of": 43,
      "values": [
        {
          "value": false,
          "count": 3
        },
        {
          "value": true,
          "count": 8
        }
      ]
    },
    {
      "id": "offering.harness_compatibility",
      "label": "Harness compatibility",
      "definition": "The registered harnesses whose own documentation names this provider as a supported backend for this model. Absence means undocumented.",
      "subject": "offering",
      "value_type": "set",
      "unit": null,
      "unit_definition": null,
      "operators": [
        "in",
        "not in",
        "known"
      ],
      "objective": false,
      "preference": null,
      "risk": "capability",
      "computed_by": null,
      "known": 0,
      "of": 43,
      "values": []
    }
  ],
  "benchmarks": [
    {
      "id": "aime_2026",
      "name": "AIME 2026",
      "unit": "percent",
      "higher_is_better": true,
      "models": 9,
      "independent_models": 9,
      "range": {
        "min": 90.0,
        "max": 99.17
      },
      "domains": [
        {
          "id": "maths",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arc_agi_2",
      "name": "ARC-AGI-2",
      "unit": "percent",
      "higher_is_better": true,
      "models": 1,
      "independent_models": 0,
      "range": {
        "min": 95.0,
        "max": 95.0
      },
      "domains": [
        {
          "id": "reasoning",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_elo_coding",
      "name": "Arena Elo — Coding",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 5,
      "independent_models": 5,
      "range": {
        "min": 1488.58,
        "max": 1535.27
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "software_engineering",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_elo_overall",
      "name": "Arena Elo — Overall (Text)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 7,
      "independent_models": 7,
      "range": {
        "min": 1443.72,
        "max": 1507.58
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_elo_style_control",
      "name": "Arena Elo — Style Control",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 20,
      "independent_models": 20,
      "range": {
        "min": 1256.13,
        "max": 1505.68
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_sc_business",
      "name": "Arena Business, Management and Financial Operations (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1455.48,
        "max": 1507.82
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "finance",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_chinese",
      "name": "Arena Chinese (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 19,
      "independent_models": 19,
      "range": {
        "min": 1493.3,
        "max": 1592.17
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_coding",
      "name": "Arena Coding (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1501.5,
        "max": 1552.39
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "software_engineering",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_creative_writing",
      "name": "Arena Creative Writing (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1430.7,
        "max": 1504.13
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "writing",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_document",
      "name": "Arena Document",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 14,
      "independent_models": 14,
      "range": {
        "min": 1443.92,
        "max": 1516.27
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "vision_documents",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_english",
      "name": "Arena English (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 20,
      "independent_models": 20,
      "range": {
        "min": 1466.79,
        "max": 1513.7
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_sc_expert",
      "name": "Arena Expert (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1472.21,
        "max": 1548.48
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_sc_factuality",
      "name": "Arena Text Factuality",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 20,
      "independent_models": 20,
      "range": {
        "min": 1454.79,
        "max": 1500.73
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "reasoning",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_french",
      "name": "Arena French (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 17,
      "independent_models": 17,
      "range": {
        "min": 1473.05,
        "max": 1531.84
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_german",
      "name": "Arena German (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 15,
      "independent_models": 15,
      "range": {
        "min": 1456.74,
        "max": 1543.7
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_hard_prompts",
      "name": "Arena Hard Prompts (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1479.8,
        "max": 1533.13
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_sc_industry_entertainment_sports_media",
      "name": "Arena Entertainment, Sports and Media (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 20,
      "independent_models": 20,
      "range": {
        "min": 1427.41,
        "max": 1495.88
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "writing",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_industry_mathematical",
      "name": "Arena Mathematical Occupations (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 19,
      "independent_models": 19,
      "range": {
        "min": 1460.32,
        "max": 1527.58
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "maths",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_industry_software_it_services",
      "name": "Arena Software and IT Services (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 20,
      "independent_models": 20,
      "range": {
        "min": 1491.76,
        "max": 1540.9
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "software_engineering",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_instruction_following",
      "name": "Arena Instruction Following (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1452.55,
        "max": 1513.49
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_sc_japanese",
      "name": "Arena Japanese (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 14,
      "independent_models": 14,
      "range": {
        "min": 1442.95,
        "max": 1523.48
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_korean",
      "name": "Arena Korean (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 15,
      "independent_models": 15,
      "range": {
        "min": 1418.01,
        "max": 1510.64
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_legal",
      "name": "Arena Legal and Government (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1461.67,
        "max": 1512.36
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "legal",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_longer_query",
      "name": "Arena Longer Query (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1471.54,
        "max": 1524.11
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_sc_math",
      "name": "Arena Math (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 16,
      "independent_models": 16,
      "range": {
        "min": 1444.63,
        "max": 1526.28
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "maths",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_medicine",
      "name": "Arena Medicine and Healthcare (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1470.9,
        "max": 1519.76
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "medical",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_multi_turn",
      "name": "Arena Multi-Turn (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1459.44,
        "max": 1517.56
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "arena_sc_non_english",
      "name": "Arena Non-English (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1444.72,
        "max": 1495.98
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_polish",
      "name": "Arena Polish (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 15,
      "independent_models": 15,
      "range": {
        "min": 1445.22,
        "max": 1510.24
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_russian",
      "name": "Arena Russian (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 20,
      "independent_models": 20,
      "range": {
        "min": 1446.5,
        "max": 1518.83
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_science",
      "name": "Arena Life, Physical and Social Science (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1477.44,
        "max": 1528.15
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "engineering_stem",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_spanish",
      "name": "Arena Spanish (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 15,
      "independent_models": 15,
      "range": {
        "min": 1445.46,
        "max": 1514.49
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "multilingual",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_vision",
      "name": "Arena Vision (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 12,
      "independent_models": 12,
      "range": {
        "min": 1262.37,
        "max": 1309.5
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "vision_documents",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_vision_diagram",
      "name": "Arena Vision Diagram (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 14,
      "independent_models": 14,
      "range": {
        "min": 1292.45,
        "max": 1355.26
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "vision_documents",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_vision_homework",
      "name": "Arena Vision Homework (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 13,
      "independent_models": 13,
      "range": {
        "min": 1280.78,
        "max": 1343.28
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "vision_documents",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_vision_ocr",
      "name": "Arena Vision OCR (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 14,
      "independent_models": 14,
      "range": {
        "min": 1277.67,
        "max": 1325.36
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "vision_documents",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_sc_writing",
      "name": "Arena Writing, Literature and Language (style control)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 18,
      "independent_models": 18,
      "range": {
        "min": 1438.36,
        "max": 1511.45
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "writing",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "arena_webdev",
      "name": "Arena WebDev (Code Arena)",
      "unit": "Arena score (Elo scale)",
      "higher_is_better": true,
      "models": 23,
      "independent_models": 23,
      "range": {
        "min": 1392.0,
        "max": 1818.41
      },
      "domains": [
        {
          "id": "chat_preference",
          "directness": "direct"
        },
        {
          "id": "software_engineering",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "browsecomp",
      "name": "BrowseComp",
      "unit": "percent",
      "higher_is_better": true,
      "models": 1,
      "independent_models": 0,
      "range": {
        "min": 91.5,
        "max": 91.5
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "cursorbench_4",
      "name": "CursorBench 4.0",
      "unit": "percent",
      "higher_is_better": true,
      "models": 8,
      "independent_models": 7,
      "range": {
        "min": 39.6,
        "max": 57.8
      },
      "domains": [
        {
          "id": "software_engineering",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "deepswe_v1_1",
      "name": "DeepSWE v1.1",
      "unit": "percent",
      "higher_is_better": true,
      "models": 12,
      "independent_models": 12,
      "range": {
        "min": 11.73,
        "max": 73.83
      },
      "domains": [
        {
          "id": "software_engineering",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "finance_benchmark_v2",
      "name": "Finance Benchmark v2",
      "unit": "percent",
      "higher_is_better": true,
      "models": 8,
      "independent_models": 8,
      "range": {
        "min": 63.0137,
        "max": 93.1507
      },
      "domains": [
        {
          "id": "finance",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "frontiercode_v1_1",
      "name": "FrontierCode 1.1",
      "unit": "percent",
      "higher_is_better": true,
      "models": 14,
      "independent_models": 13,
      "range": {
        "min": 17.6,
        "max": 53.5
      },
      "domains": [
        {
          "id": "software_engineering",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "frontiermath_tiers_1_3_v2",
      "name": "FrontierMath Tiers 1-3 (v2)",
      "unit": "percent",
      "higher_is_better": true,
      "models": 17,
      "independent_models": 17,
      "range": {
        "min": 45.26,
        "max": 93.68
      },
      "domains": [
        {
          "id": "maths",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "gpqa_diamond",
      "name": "GPQA Diamond",
      "unit": "percent",
      "higher_is_better": true,
      "models": 19,
      "independent_models": 18,
      "range": {
        "min": 75.76,
        "max": 95.77
      },
      "domains": [
        {
          "id": "engineering_stem",
          "directness": "direct"
        },
        {
          "id": "reasoning",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "hle",
      "name": "Humanity's Last Exam",
      "unit": "percent",
      "higher_is_better": true,
      "models": 8,
      "independent_models": 8,
      "range": {
        "min": 34.44,
        "max": 54.8
      },
      "domains": [
        {
          "id": "engineering_stem",
          "directness": "direct"
        },
        {
          "id": "maths",
          "directness": "proxy"
        },
        {
          "id": "reasoning",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "hle_tools",
      "name": "Humanity's Last Exam (with tools)",
      "unit": "percent",
      "higher_is_better": true,
      "models": 1,
      "independent_models": 0,
      "range": {
        "min": 57.2,
        "max": 57.2
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "proxy"
        },
        {
          "id": "engineering_stem",
          "directness": "direct"
        },
        {
          "id": "maths",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "metr_time_horizon_50",
      "name": "METR time horizon (50% success, TH 1.1)",
      "unit": "minutes",
      "higher_is_better": true,
      "models": 3,
      "independent_models": 3,
      "range": {
        "min": 341.735276,
        "max": 718.80683
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        },
        {
          "id": "software_engineering",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "metr_time_horizon_80",
      "name": "METR time horizon (80% success, TH 1.1)",
      "unit": "minutes",
      "higher_is_better": true,
      "models": 3,
      "independent_models": 3,
      "range": {
        "min": 53.877851,
        "max": 89.801503
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        },
        {
          "id": "software_engineering",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "mteb_eng_v2",
      "name": "MTEB(eng, v2), mean over tasks",
      "unit": "percent",
      "higher_is_better": true,
      "models": 4,
      "independent_models": 4,
      "range": {
        "min": 74.76,
        "max": 75.98
      },
      "domains": [
        {
          "id": "retrieval",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "mteb_multilingual_v2",
      "name": "MTEB(Multilingual, v2), mean over tasks",
      "unit": "percent",
      "higher_is_better": true,
      "models": 2,
      "independent_models": 2,
      "range": {
        "min": 72.32,
        "max": 74.27
      },
      "domains": [
        {
          "id": "multilingual",
          "directness": "proxy"
        },
        {
          "id": "retrieval",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "mteb_v2_reranking",
      "name": "MTEB(eng, v2) Reranking",
      "unit": "percent",
      "higher_is_better": true,
      "models": 2,
      "independent_models": 2,
      "range": {
        "min": 48.27,
        "max": 49.2
      },
      "domains": [
        {
          "id": "retrieval",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "mteb_v2_retrieval",
      "name": "MTEB(eng, v2) Retrieval",
      "unit": "percent",
      "higher_is_better": true,
      "models": 2,
      "independent_models": 2,
      "range": {
        "min": 67.12,
        "max": 70.0
      },
      "domains": [
        {
          "id": "retrieval",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "osworld_2",
      "name": "OSWorld 2.0",
      "unit": "percent",
      "higher_is_better": true,
      "models": 3,
      "independent_models": 3,
      "range": {
        "min": 18.2,
        "max": 31.4
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        },
        {
          "id": "vision_documents",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "swe_bench_multilingual",
      "name": "SWE-bench Multilingual",
      "unit": "percent",
      "higher_is_better": true,
      "models": 1,
      "independent_models": 1,
      "range": {
        "min": 72.0,
        "max": 72.0
      },
      "domains": [
        {
          "id": "software_engineering",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "swe_bench_pro",
      "name": "SWE-bench Pro",
      "unit": "percent",
      "higher_is_better": true,
      "models": 5,
      "independent_models": 5,
      "range": {
        "min": 46.1,
        "max": 61.5
      },
      "domains": [
        {
          "id": "software_engineering",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "swe_bench_verified",
      "name": "SWE-bench Verified",
      "unit": "percent",
      "higher_is_better": true,
      "models": 7,
      "independent_models": 7,
      "range": {
        "min": 75.6,
        "max": 83.47
      },
      "domains": [
        {
          "id": "software_engineering",
          "directness": "direct"
        }
      ]
    },
    {
      "id": "tau3_banking",
      "name": "τ³-Banking (τ-Knowledge)",
      "unit": "percent",
      "higher_is_better": true,
      "models": 9,
      "independent_models": 9,
      "range": {
        "min": 26.03,
        "max": 48.71
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        },
        {
          "id": "finance",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "terminal_bench_science",
      "name": "Terminal-Bench-Science",
      "unit": "percent",
      "higher_is_better": true,
      "models": 1,
      "independent_models": 0,
      "range": {
        "min": 64.6,
        "max": 64.6
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        },
        {
          "id": "engineering_stem",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "terminal_bench_v4_0",
      "name": "Terminal-Bench 4.0",
      "unit": "percent",
      "higher_is_better": true,
      "models": 10,
      "independent_models": 9,
      "range": {
        "min": 11.21,
        "max": 70.6
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        },
        {
          "id": "software_engineering",
          "directness": "proxy"
        }
      ]
    },
    {
      "id": "vending_bench_2",
      "name": "Vending-Bench 2",
      "unit": "USD",
      "higher_is_better": true,
      "models": 16,
      "independent_models": 16,
      "range": {
        "min": 911.21,
        "max": 15514.7
      },
      "domains": [
        {
          "id": "agentic_tool_use",
          "directness": "direct"
        },
        {
          "id": "finance",
          "directness": "proxy"
        }
      ]
    }
  ],
  "domains": [
    {
      "id": "software_engineering",
      "name": "Software engineering",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 26,
      "estimate_benchmarks": [
        "arena_elo_coding",
        "arena_sc_coding",
        "arena_sc_industry_software_it_services",
        "arena_webdev",
        "cursorbench_4",
        "deepswe_v1_1",
        "frontiercode_v1_1",
        "metr_time_horizon_50",
        "metr_time_horizon_80",
        "swe_bench_pro",
        "swe_bench_verified",
        "terminal_bench_v4_0"
      ],
      "direct_models": 23,
      "default_benchmark": "swe_bench_pro",
      "benchmarks": [
        "swe_bench_pro",
        "frontiercode_v1_1",
        "deepswe_v1_1",
        "cursorbench_4",
        "swe_bench_verified",
        "swe_bench_multilingual",
        "arena_webdev",
        "arena_sc_industry_software_it_services",
        "arena_sc_coding",
        "terminal_bench_v4_0",
        "arena_elo_coding",
        "metr_time_horizon_50",
        "metr_time_horizon_80"
      ]
    },
    {
      "id": "engineering_stem",
      "name": "Engineering and STEM",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 19,
      "estimate_benchmarks": [
        "arena_sc_science",
        "gpqa_diamond",
        "hle"
      ],
      "direct_models": 20,
      "default_benchmark": null,
      "benchmarks": [
        "gpqa_diamond",
        "hle",
        "hle_tools",
        "arena_sc_science",
        "terminal_bench_science"
      ]
    },
    {
      "id": "maths",
      "name": "Maths",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 20,
      "estimate_benchmarks": [
        "aime_2026",
        "arena_sc_industry_mathematical",
        "arena_sc_math",
        "frontiermath_tiers_1_3_v2",
        "hle"
      ],
      "direct_models": 17,
      "default_benchmark": null,
      "benchmarks": [
        "frontiermath_tiers_1_3_v2",
        "aime_2026",
        "arena_sc_industry_mathematical",
        "arena_sc_math",
        "hle",
        "hle_tools"
      ]
    },
    {
      "id": "reasoning",
      "name": "General reasoning",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 21,
      "estimate_benchmarks": [
        "arena_sc_factuality",
        "gpqa_diamond",
        "hle"
      ],
      "direct_models": 1,
      "default_benchmark": null,
      "benchmarks": [
        "arc_agi_2",
        "arena_sc_factuality",
        "gpqa_diamond",
        "hle"
      ]
    },
    {
      "id": "legal",
      "name": "Legal",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 18,
      "estimate_benchmarks": [
        "arena_sc_legal"
      ],
      "direct_models": 0,
      "default_benchmark": null,
      "benchmarks": [
        "arena_sc_legal"
      ]
    },
    {
      "id": "medical",
      "name": "Medical",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 18,
      "estimate_benchmarks": [
        "arena_sc_medicine"
      ],
      "direct_models": 0,
      "default_benchmark": null,
      "benchmarks": [
        "arena_sc_medicine"
      ]
    },
    {
      "id": "finance",
      "name": "Finance",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 18,
      "estimate_benchmarks": [
        "arena_sc_business",
        "finance_benchmark_v2",
        "tau3_banking",
        "vending_bench_2"
      ],
      "direct_models": 8,
      "default_benchmark": null,
      "benchmarks": [
        "finance_benchmark_v2",
        "arena_sc_business",
        "vending_bench_2",
        "tau3_banking"
      ]
    },
    {
      "id": "writing",
      "name": "Writing",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 20,
      "estimate_benchmarks": [
        "arena_sc_creative_writing",
        "arena_sc_industry_entertainment_sports_media",
        "arena_sc_writing"
      ],
      "direct_models": 0,
      "default_benchmark": null,
      "benchmarks": [
        "arena_sc_industry_entertainment_sports_media",
        "arena_sc_creative_writing",
        "arena_sc_writing"
      ]
    },
    {
      "id": "agentic_tool_use",
      "name": "Agentic and tool use",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 19,
      "estimate_benchmarks": [
        "metr_time_horizon_50",
        "metr_time_horizon_80",
        "osworld_2",
        "tau3_banking",
        "terminal_bench_v4_0",
        "vending_bench_2"
      ],
      "direct_models": 19,
      "default_benchmark": null,
      "benchmarks": [
        "vending_bench_2",
        "terminal_bench_v4_0",
        "tau3_banking",
        "metr_time_horizon_50",
        "metr_time_horizon_80",
        "osworld_2",
        "browsecomp",
        "terminal_bench_science",
        "hle_tools"
      ]
    },
    {
      "id": "vision_documents",
      "name": "Vision and documents",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 14,
      "estimate_benchmarks": [
        "arena_sc_document",
        "arena_sc_vision",
        "arena_sc_vision_diagram",
        "arena_sc_vision_homework",
        "arena_sc_vision_ocr",
        "osworld_2"
      ],
      "direct_models": 0,
      "default_benchmark": null,
      "benchmarks": [
        "arena_sc_document",
        "arena_sc_vision_diagram",
        "arena_sc_vision_ocr",
        "arena_sc_vision_homework",
        "arena_sc_vision",
        "osworld_2"
      ]
    },
    {
      "id": "multilingual",
      "name": "Multilingual",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 22,
      "estimate_benchmarks": [
        "arena_sc_chinese",
        "arena_sc_french",
        "arena_sc_german",
        "arena_sc_japanese",
        "arena_sc_korean",
        "arena_sc_non_english",
        "arena_sc_polish",
        "arena_sc_russian",
        "arena_sc_spanish",
        "mteb_multilingual_v2"
      ],
      "direct_models": 0,
      "default_benchmark": null,
      "benchmarks": [
        "arena_sc_russian",
        "arena_sc_chinese",
        "arena_sc_non_english",
        "arena_sc_french",
        "arena_sc_german",
        "arena_sc_korean",
        "arena_sc_polish",
        "arena_sc_spanish",
        "arena_sc_japanese",
        "mteb_multilingual_v2"
      ]
    },
    {
      "id": "chat_preference",
      "name": "Chat and preference",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 27,
      "estimate_benchmarks": [
        "arena_elo_coding",
        "arena_elo_overall",
        "arena_elo_style_control",
        "arena_sc_business",
        "arena_sc_chinese",
        "arena_sc_coding",
        "arena_sc_creative_writing",
        "arena_sc_document",
        "arena_sc_english",
        "arena_sc_expert",
        "arena_sc_factuality",
        "arena_sc_french",
        "arena_sc_german",
        "arena_sc_industry_entertainment_sports_media",
        "arena_webdev"
      ],
      "direct_models": 27,
      "default_benchmark": "arena_elo_style_control",
      "benchmarks": [
        "arena_elo_style_control",
        "arena_webdev",
        "arena_sc_english",
        "arena_sc_factuality",
        "arena_sc_industry_entertainment_sports_media",
        "arena_sc_industry_software_it_services",
        "arena_sc_russian",
        "arena_sc_chinese",
        "arena_sc_industry_mathematical",
        "arena_sc_business",
        "arena_sc_coding",
        "arena_sc_creative_writing",
        "arena_sc_expert",
        "arena_sc_hard_prompts",
        "arena_sc_instruction_following",
        "arena_sc_legal",
        "arena_sc_longer_query",
        "arena_sc_medicine",
        "arena_sc_multi_turn",
        "arena_sc_non_english",
        "arena_sc_science",
        "arena_sc_writing",
        "arena_sc_french",
        "arena_sc_math",
        "arena_sc_german",
        "arena_sc_korean",
        "arena_sc_polish",
        "arena_sc_spanish",
        "arena_sc_document",
        "arena_sc_japanese",
        "arena_sc_vision_diagram",
        "arena_sc_vision_ocr",
        "arena_sc_vision_homework",
        "arena_sc_vision",
        "arena_elo_overall",
        "arena_elo_coding"
      ]
    },
    {
      "id": "retrieval",
      "name": "Retrieval",
      "proxy_only": false,
      "default_basis": "capability_estimate",
      "estimate_models": 8,
      "estimate_benchmarks": [
        "mteb_eng_v2",
        "mteb_multilingual_v2",
        "mteb_v2_reranking",
        "mteb_v2_retrieval"
      ],
      "direct_models": 4,
      "default_benchmark": "mteb_v2_retrieval",
      "benchmarks": [
        "mteb_v2_retrieval",
        "mteb_v2_reranking",
        "mteb_eng_v2",
        "mteb_multilingual_v2"
      ]
    }
  ],
  "models": {
    "anthropic/claude-fable-5": {
      "display_name": "Claude Fable 5",
      "lab": "anthropic",
      "lab_name": "Anthropic",
      "class": "text-generator"
    },
    "anthropic/claude-fable-5-1": {
      "display_name": "Claude Fable 5.1",
      "lab": "anthropic",
      "lab_name": "Anthropic",
      "class": "text-generator"
    },
    "anthropic/claude-opus-4-6": {
      "display_name": "Claude Opus 4.6",
      "lab": "anthropic",
      "lab_name": "Anthropic",
      "class": "text-generator"
    },
    "anthropic/claude-opus-4-7": {
      "display_name": "Claude Opus 4.7",
      "lab": "anthropic",
      "lab_name": "Anthropic",
      "class": "text-generator"
    },
    "anthropic/claude-opus-5": {
      "display_name": "Claude Opus 5",
      "lab": "anthropic",
      "lab_name": "Anthropic",
      "class": "text-generator"
    },
    "anthropic/claude-opus-5-5": {
      "display_name": "Claude Opus 5.5",
      "lab": "anthropic",
      "lab_name": "Anthropic",
      "class": "text-generator"
    },
    "anthropic/claude-sonnet-5-5": {
      "display_name": "Claude Sonnet 5.5",
      "lab": "anthropic",
      "lab_name": "Anthropic",
      "class": "text-generator"
    },
    "bytedance/seed1-5-embedding": {
      "display_name": "Seed1.5-Embedding",
      "lab": "bytedance",
      "lab_name": "ByteDance Seed",
      "class": "vectoriser"
    },
    "deepseek/deepseek-flash": {
      "display_name": "DeepSeek V4.1 Flash",
      "lab": "deepseek",
      "lab_name": "DeepSeek",
      "class": "text-generator"
    },
    "deepseek/deepseek-v3-1": {
      "display_name": "DeepSeek V3.1",
      "lab": "deepseek",
      "lab_name": "DeepSeek",
      "class": "text-generator"
    },
    "deepseek/deepseek-v4-pro": {
      "display_name": "DeepSeek V4 Pro",
      "lab": "deepseek",
      "lab_name": "DeepSeek",
      "class": "text-generator"
    },
    "google/gemini-2-5-flash": {
      "display_name": "Gemini 2.5 Flash",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemini-3-1-pro-preview": {
      "display_name": "Gemini 3.1 Pro Preview",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemini-3-5-flash": {
      "display_name": "Gemini 3.5 Flash",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemini-3-7-flash": {
      "display_name": "Gemini 3.7 Flash",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemini-3-8-flash": {
      "display_name": "Gemini 3.8 Flash",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemma-4-26b-a4b-it": {
      "display_name": "gemma 4 26B A4B it",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemma-4-31b-it": {
      "display_name": "gemma 4 31B it",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemma-4-e2b-it": {
      "display_name": "gemma 4 E2B it",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "google/gemma-4-e4b-it": {
      "display_name": "gemma 4 E4B it",
      "lab": "google",
      "lab_name": "Google DeepMind",
      "class": "text-generator"
    },
    "jcorners/ingot-8b-r3": {
      "display_name": "Ingot 8B R3",
      "lab": "jcorners",
      "lab_name": "Voxell",
      "class": "vectoriser"
    },
    "kingsoft/qzhou-embedding": {
      "display_name": "QZhou-Embedding",
      "lab": "kingsoft",
      "lab_name": "Kingsoft",
      "class": "vectoriser"
    },
    "meta/muse-spark": {
      "display_name": "Muse Spark",
      "lab": "meta",
      "lab_name": "Meta",
      "class": "text-generator"
    },
    "meta/muse-spark-1-1": {
      "display_name": "Muse Spark 1.1",
      "lab": "meta",
      "lab_name": "Meta",
      "class": "text-generator"
    },
    "meta/muse-spark-1-3": {
      "display_name": "Muse Spark 1.3",
      "lab": "meta",
      "lab_name": "Meta",
      "class": "text-generator"
    },
    "microsoft/harrier-oss-v1-27b": {
      "display_name": "harrier-oss-v1-27b",
      "lab": "microsoft",
      "lab_name": "Microsoft",
      "class": "vectoriser"
    },
    "microsoft/phi-4": {
      "display_name": "phi 4",
      "lab": "microsoft",
      "lab_name": "Microsoft",
      "class": "text-generator"
    },
    "moonshot/kimi-k2-6": {
      "display_name": "Kimi K2.6",
      "lab": "moonshot",
      "lab_name": "Moonshot AI",
      "class": "text-generator"
    },
    "moonshot/kimi-k3": {
      "display_name": "Kimi K3",
      "lab": "moonshot",
      "lab_name": "Moonshot AI",
      "class": "text-generator"
    },
    "openai/gpt-5-4": {
      "display_name": "GPT-5.4",
      "lab": "openai",
      "lab_name": "OpenAI",
      "class": "text-generator"
    },
    "openai/gpt-5-6-sol": {
      "display_name": "GPT-5.6 Sol",
      "lab": "openai",
      "lab_name": "OpenAI",
      "class": "text-generator"
    },
    "openai/gpt-6-astra": {
      "display_name": "GPT-6 Astra",
      "lab": "openai",
      "lab_name": "OpenAI",
      "class": "text-generator"
    },
    "openai/gpt-6-luna": {
      "display_name": "GPT-6 Luna",
      "lab": "openai",
      "lab_name": "OpenAI",
      "class": "text-generator"
    },
    "openai/gpt-6-sol": {
      "display_name": "GPT-6 Sol",
      "lab": "openai",
      "lab_name": "OpenAI",
      "class": "text-generator"
    },
    "querit/querit": {
      "display_name": "Querit",
      "lab": "querit",
      "lab_name": "Querit",
      "class": "orderer"
    },
    "querit/querit-4b": {
      "display_name": "Querit-4B",
      "lab": "querit",
      "lab_name": "Querit",
      "class": "orderer"
    },
    "qwen/qwen3-8-flash-next": {
      "display_name": "Qwen3.8-Flash-Next",
      "lab": "qwen",
      "lab_name": "Alibaba Cloud",
      "class": "text-generator"
    },
    "qwen/qwen3-8-max-0902": {
      "display_name": "Qwen3.8 Max 0902",
      "lab": "qwen",
      "lab_name": "Alibaba / Qwen Team",
      "class": "text-generator"
    },
    "qwen/qwen3-embedding-8b": {
      "display_name": "Qwen3 Embedding 8B",
      "lab": "qwen",
      "lab_name": "Alibaba / Qwen Team",
      "class": "vectoriser"
    },
    "tencent/kalm-embedding-gemma3-12b-2511": {
      "display_name": "KaLM Embedding Gemma3 12B 2511",
      "lab": "tencent",
      "lab_name": "Tencent",
      "class": "vectoriser"
    },
    "typesafe/jev-1-13": {
      "display_name": "Jev 1.13",
      "lab": "typesafe",
      "lab_name": "TypeSafe AI",
      "class": "decider"
    },
    "xai/grok-4-7": {
      "display_name": "Grok 4.7",
      "lab": "xai",
      "lab_name": "xAI",
      "class": "text-generator"
    },
    "zhipu/glm-5-2": {
      "display_name": "GLM-5.2",
      "lab": "zhipu",
      "lab_name": "Z.ai (Zhipu AI)",
      "class": "text-generator"
    },
    "zhipu/glm-5-3": {
      "display_name": "GLM-5.3",
      "lab": "zhipu",
      "lab_name": "Z.ai (Zhipu AI)",
      "class": "text-generator"
    }
  },
  "providers": {
    "openai": "OpenAI API",
    "anthropic": "Anthropic API",
    "google-gemini-api": "Gemini API (Google AI Studio)",
    "xai": "xAI API",
    "mistral": "Mistral AI La Plateforme",
    "cohere": "Cohere API",
    "deepseek": "DeepSeek Open Platform",
    "alibaba-model-studio": "Alibaba Cloud Model Studio",
    "moonshot": "Kimi Open Platform (Moonshot AI)",
    "zai": "Z.ai API",
    "minimax": "MiniMax API Platform",
    "ai21": "AI21 Studio",
    "meta-model-api": "Meta Model API",
    "typesafe": "TypeSafe API",
    "aws-bedrock": "Amazon Bedrock",
    "google-vertex-ai": "Vertex AI (Google Cloud)",
    "azure-ai-foundry": "Azure AI Foundry",
    "groq": "GroqCloud",
    "together-ai": "Together AI",
    "fireworks-ai": "Fireworks AI",
    "deepinfra": "DeepInfra",
    "cerebras": "Cerebras Inference",
    "sambanova": "SambaNova Cloud",
    "nvidia-nim": "NVIDIA NIM (build.nvidia.com)",
    "replicate": "Replicate",
    "openrouter": "OpenRouter"
  },
  "coverage": {
    "as_of": "2026-09-29",
    "models": 44,
    "verified": 38,
    "classes": [
      {
        "id": "actor",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "analyser",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "decider",
        "models": 1,
        "verified": 0,
        "domains": []
      },
      {
        "id": "forecaster",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "labeller",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "media-generator",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "orderer",
        "models": 2,
        "verified": 2,
        "domains": [
          {
            "id": "retrieval",
            "verified": 2
          }
        ]
      },
      {
        "id": "scorer",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "simulator",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "text-generator",
        "models": 35,
        "verified": 30,
        "domains": [
          {
            "id": "chat_preference",
            "verified": 27
          },
          {
            "id": "software_engineering",
            "verified": 26
          },
          {
            "id": "reasoning",
            "verified": 22
          },
          {
            "id": "engineering_stem",
            "verified": 20
          },
          {
            "id": "maths",
            "verified": 20
          },
          {
            "id": "multilingual",
            "verified": 20
          },
          {
            "id": "writing",
            "verified": 20
          },
          {
            "id": "agentic_tool_use",
            "verified": 19
          },
          {
            "id": "finance",
            "verified": 18
          },
          {
            "id": "legal",
            "verified": 18
          },
          {
            "id": "medical",
            "verified": 18
          },
          {
            "id": "vision_documents",
            "verified": 14
          }
        ]
      },
      {
        "id": "transcriber",
        "models": 0,
        "verified": 0,
        "domains": []
      },
      {
        "id": "vectoriser",
        "models": 6,
        "verified": 6,
        "domains": [
          {
            "id": "retrieval",
            "verified": 6
          },
          {
            "id": "multilingual",
            "verified": 2
          }
        ]
      }
    ],
    "domains": [
      {
        "id": "software_engineering",
        "name": "Software engineering",
        "verified": 26,
        "direct": 23
      },
      {
        "id": "engineering_stem",
        "name": "Engineering and STEM",
        "verified": 20,
        "direct": 20
      },
      {
        "id": "maths",
        "name": "Maths",
        "verified": 20,
        "direct": 17
      },
      {
        "id": "reasoning",
        "name": "General reasoning",
        "verified": 22,
        "direct": 1
      },
      {
        "id": "legal",
        "name": "Legal",
        "verified": 18,
        "direct": 0
      },
      {
        "id": "medical",
        "name": "Medical",
        "verified": 18,
        "direct": 0
      },
      {
        "id": "finance",
        "name": "Finance",
        "verified": 18,
        "direct": 8
      },
      {
        "id": "writing",
        "name": "Writing",
        "verified": 20,
        "direct": 0
      },
      {
        "id": "marketing_seo",
        "name": "Marketing and SEO",
        "verified": 0,
        "direct": 0
      },
      {
        "id": "agentic_tool_use",
        "name": "Agentic and tool use",
        "verified": 19,
        "direct": 19
      },
      {
        "id": "vision_documents",
        "name": "Vision and documents",
        "verified": 14,
        "direct": 0
      },
      {
        "id": "multilingual",
        "name": "Multilingual",
        "verified": 22,
        "direct": 0
      },
      {
        "id": "chat_preference",
        "name": "Chat and preference",
        "verified": 27,
        "direct": 27
      },
      {
        "id": "retrieval",
        "name": "Retrieval",
        "verified": 8,
        "direct": 4
      }
    ]
  },
  "templates": [
    {
      "id": "budget-coding",
      "name": "Coding agent on a budget",
      "purpose": "Find an active coding model with enough context while keeping one task under $0.25.",
      "where": [
        {
          "condition": "model.class = text-generator",
          "reason": "Coding work needs a model that generates text and code."
        },
        {
          "condition": "model.lifecycle = active",
          "reason": "An active model is still supported by its lab or provider."
        },
        {
          "condition": "model.context_window >= 200000",
          "reason": "A large repository needs room for code, tests, and instructions."
        },
        {
          "condition": "offering.cost_per_task <= 0.25",
          "reason": "The offering must stay within the per-task budget."
        }
      ],
      "weights": {
        "software_engineering": {
          "weight": 0.6,
          "reason": "Prefer stronger evidence on real software-engineering tasks."
        },
        "-offering.cost_per_task": {
          "weight": 0.4,
          "reason": "Among qualifying offerings, prefer the cheaper task."
        }
      },
      "needs": {
        "classes": [
          "text-generator"
        ],
        "domains": [
          "software_engineering"
        ]
      },
      "teaches": "Use Must for non-negotiable context and budget limits, then Prefer to trade coding strength against cost.",
      "spec": {
        "spec_version": 1,
        "where": [
          "model.class = text-generator",
          "model.lifecycle = active",
          "model.context_window >= 200000",
          "offering.cost_per_task <= 0.25"
        ],
        "optimize": {
          "weights": {
            "software_engineering": 0.6,
            "-offering.cost_per_task": 0.4
          }
        }
      },
      "available": true,
      "unavailable_reason": null
    },
    {
      "id": "private-self-host",
      "name": "Private assistant you host yourself",
      "purpose": "Find a commercially usable open-weights assistant that you can operate yourself.",
      "where": [
        {
          "condition": "model.class = text-generator",
          "reason": "A conversational assistant needs a text-generating model."
        },
        {
          "condition": "model.weights_openness = open_weights",
          "reason": "Self-hosting requires downloadable weights; add a model.fits_hardware Must for the device you own."
        },
        {
          "condition": "licence.commercial_use in {permitted, permitted_with_conditions}",
          "reason": "The licence must allow commercial use, including use with stated conditions."
        }
      ],
      "weights": {
        "chat_preference": {
          "weight": 0.7,
          "reason": "Prefer assistants whose open-ended responses people rate more highly."
        },
        "-offering.cost_per_task": {
          "weight": 0.3,
          "reason": "Prefer lower task cost when the snapshot has an offering price."
        }
      },
      "needs": {
        "classes": [
          "text-generator"
        ],
        "domains": [
          "chat_preference"
        ]
      },
      "teaches": "Use Must for deployment and licence requirements, then Prefer to balance assistant quality and cost.",
      "spec": {
        "spec_version": 1,
        "where": [
          "model.class = text-generator",
          "model.weights_openness = open_weights",
          "licence.commercial_use in {permitted, permitted_with_conditions}"
        ],
        "optimize": {
          "weights": {
            "chat_preference": 0.7,
            "-offering.cost_per_task": 0.3
          }
        }
      },
      "available": true,
      "unavailable_reason": null
    },
    {
      "id": "regulated-data",
      "name": "Regulated data",
      "purpose": "Find an offering with the data-handling commitments needed for regulated workloads.",
      "where": [
        {
          "condition": "offering.data.trains_on_customer_data = false",
          "reason": "Customer prompts and outputs must not be available for model training."
        },
        {
          "condition": "offering.data.zero_retention = true",
          "reason": "The provider must offer a mode that stores neither prompts nor outputs."
        },
        {
          "condition": "offering.attestation.baa = true",
          "reason": "The provider must offer a HIPAA business associate agreement for the service."
        }
      ],
      "weights": {
        "chat_preference": {
          "weight": 0.6,
          "reason": "Prefer the stronger general assistant among compliant offerings."
        },
        "-offering.cost_per_task": {
          "weight": 0.4,
          "reason": "Prefer the lower task cost after the compliance gates pass."
        }
      },
      "needs": {
        "classes": [],
        "domains": [
          "chat_preference"
        ]
      },
      "teaches": "Governance promises are Musts because a high score cannot compensate for a missing data commitment.",
      "spec": {
        "spec_version": 1,
        "where": [
          "offering.data.trains_on_customer_data = false",
          "offering.data.zero_retention = true",
          "offering.attestation.baa = true"
        ],
        "optimize": {
          "weights": {
            "chat_preference": 0.6,
            "-offering.cost_per_task": 0.4
          }
        }
      },
      "available": true,
      "unavailable_reason": null
    },
    {
      "id": "maths",
      "name": "Maths and proofs",
      "purpose": "Find a text model with the strongest evidence on mathematical problems and proofs.",
      "where": [
        {
          "condition": "model.class = text-generator",
          "reason": "The answer must be generated as mathematical text or a proof."
        }
      ],
      "weights": {
        "maths": {
          "weight": 1.0,
          "reason": "Rank only by the maths capability estimate."
        }
      },
      "needs": {
        "classes": [
          "text-generator"
        ],
        "domains": [
          "maths"
        ]
      },
      "teaches": "Use one Must to select the right model class and one Prefer to state what best means.",
      "spec": {
        "spec_version": 1,
        "where": [
          "model.class = text-generator"
        ],
        "optimize": {
          "weights": {
            "maths": 1.0
          }
        }
      },
      "available": true,
      "unavailable_reason": null
    },
    {
      "id": "retrieval-embeddings",
      "name": "Retrieval embeddings",
      "purpose": "Find an embedding model that ranks relevant documents well for retrieval.",
      "where": [
        {
          "condition": "model.class = vectoriser",
          "reason": "Retrieval embeddings require the model class that emits vectors."
        }
      ],
      "weights": {
        "retrieval": {
          "weight": 1.0,
          "reason": "Rank only by the retrieval capability estimate."
        }
      },
      "needs": {
        "classes": [
          "vectoriser"
        ],
        "domains": [
          "retrieval"
        ]
      },
      "teaches": "Model class is a Must, while measured retrieval quality is a Prefer.",
      "spec": {
        "spec_version": 1,
        "where": [
          "model.class = vectoriser"
        ],
        "optimize": {
          "weights": {
            "retrieval": 1.0
          }
        }
      },
      "available": true,
      "unavailable_reason": null
    },
    {
      "id": "high-volume",
      "name": "High volume, good enough",
      "purpose": "Find a low-cost text model for many short assistant requests without ignoring response quality.",
      "where": [
        {
          "condition": "model.class = text-generator",
          "reason": "The workload needs generated text responses."
        }
      ],
      "weights": {
        "-offering.cost_per_task": {
          "weight": 0.7,
          "reason": "At high volume, small differences in task cost dominate total spend."
        },
        "chat_preference": {
          "weight": 0.3,
          "reason": "Keep a quality signal so cheap but poor responses do not win by cost alone."
        }
      },
      "task_tokens": {
        "input": 2000,
        "output": 500
      },
      "needs": {
        "classes": [
          "text-generator"
        ],
        "domains": [
          "chat_preference"
        ]
      },
      "teaches": "Keep the model class as a Must, then make cost the stronger Prefer without turning quality into a gate.",
      "spec": {
        "spec_version": 1,
        "where": [
          "model.class = text-generator"
        ],
        "optimize": {
          "weights": {
            "-offering.cost_per_task": 0.7,
            "chat_preference": 0.3
          }
        },
        "task_tokens": {
          "input": 2000,
          "output": 500
        }
      },
      "available": true,
      "unavailable_reason": null
    },
    {
      "id": "long-documents",
      "name": "Long documents",
      "purpose": "Find a text model that can accept million-token documents and produce strong written analysis.",
      "where": [
        {
          "condition": "model.class = text-generator",
          "reason": "The result must be generated as text."
        },
        {
          "condition": "model.context_window >= 1000000",
          "reason": "The model must accept a million-token document in one request."
        }
      ],
      "weights": {
        "writing": {
          "weight": 0.5,
          "reason": "Prefer clear, high-quality long-form writing."
        },
        "reasoning": {
          "weight": 0.3,
          "reason": "Prefer stronger multi-step analysis of the document."
        },
        "-offering.cost_per_task": {
          "weight": 0.2,
          "reason": "Use cost as a smaller tie-breaker after writing and reasoning quality."
        }
      },
      "needs": {
        "classes": [
          "text-generator"
        ],
        "domains": [
          "writing",
          "reasoning"
        ]
      },
      "teaches": "Context capacity is a Must; writing, reasoning, and cost remain trade-offs expressed as Prefers.",
      "spec": {
        "spec_version": 1,
        "where": [
          "model.class = text-generator",
          "model.context_window >= 1000000"
        ],
        "optimize": {
          "weights": {
            "writing": 0.5,
            "reasoning": 0.3,
            "-offering.cost_per_task": 0.2
          }
        }
      },
      "available": true,
      "unavailable_reason": null
    },
    {
      "id": "eu-data",
      "name": "EU-only data handling",
      "purpose": "Find an offering that runs inference in an EU member state and does not train on customer data.",
      "where": [
        {
          "condition": "offering.region in {AT, BE, BG, HR, CY, CZ, DE, DK, EE, ES, FI, FR, GR, HU, IE, IT, LT, LU, LV, MT, NL, PL, PT, RO, SE, SI, SK}",
          "reason": "Inference must run in one of the EU's 27 member-state ISO country codes (EU Publications Office Interinstitutional Style Guide, https://style-guide.europa.eu/en/content/-/isg/topic?identifier=annex-a5-list-countries-territories-currencies, checked 2026-09-27)."
        },
        {
          "condition": "offering.data.trains_on_customer_data = false",
          "reason": "Customer prompts and outputs must not be available for model training."
        }
      ],
      "weights": {
        "chat_preference": {
          "weight": 0.6,
          "reason": "Prefer the stronger general assistant among offerings that meet the data rules."
        },
        "-offering.cost_per_task": {
          "weight": 0.4,
          "reason": "Prefer lower task cost after the data-handling gates pass."
        }
      },
      "needs": {
        "classes": [],
        "domains": [
          "chat_preference"
        ]
      },
      "teaches": "Residency and training policy are Musts; quality and cost only rank offerings that pass them.",
      "spec": {
        "spec_version": 1,
        "where": [
          "offering.region in {AT, BE, BG, HR, CY, CZ, DE, DK, EE, ES, FI, FR, GR, HU, IE, IT, LT, LU, LV, MT, NL, PL, PT, RO, SE, SI, SK}",
          "offering.data.trains_on_customer_data = false"
        ],
        "optimize": {
          "weights": {
            "chat_preference": 0.6,
            "-offering.cost_per_task": 0.4
          }
        }
      },
      "available": false,
      "unavailable_reason": "No offering passes: Inference region in the EU — 0 of 43 offerings"
    }
  ]
}
