{"benchmark_ranges": {"aa_briefcase": [0.0, 100.0], "aa_lcr": [0.0, 100.0], "ai2d": [40.0, 100.0], "aider_polyglot": [0.0, 90.0], "aime_2025": [0.0, 80.0], "aime_2026": [0.0, 80.0], "alpaca_eval": [0.0, 60.0], "arc_agi_2": [0.0, 80.0], "arc_challenge": [40.0, 100.0], "arena_elo_coding": [1000.0, 1400.0], "arena_elo_hard_prompts": [1000.0, 1400.0], "arena_elo_math": [1000.0, 1400.0], "arena_elo_overall": [1000.0, 1400.0], "arena_elo_style_control": [1000.0, 1400.0], "arena_elo_vision": [1000.0, 1400.0], "automationbench_aa": [0.0, 100.0], "bbh": [20.0, 95.0], "bbq": [30.0, 100.0], "beir": [20.0, 70.0], "bioasq": [20.0, 80.0], "browsecomp": [0.0, 90.0], "chartqa": [40.0, 100.0], "charxiv_reasoning": [20.0, 95.0], "charxiv_reasoning_tools": [20.0, 95.0], "convfinqa": [20.0, 80.0], "critpt": [0.0, 100.0], "deepsearchqa": [0.0, 80.0], "docvqa": [40.0, 100.0], "fid": [1.0, 100.0], "finbench": [0.0, 100.0], "finqa": [20.0, 90.0], "flores": [10.0, 70.0], "flores_en_de": [10.0, 50.0], "flores_en_es": [10.0, 50.0], "flores_en_fr": [10.0, 50.0], "flores_en_ja": [10.0, 50.0], "flores_en_ko": [10.0, 50.0], "flores_en_zh": [10.0, 50.0], "fpb": [20.0, 90.0], "frontierscience_research": [0.0, 50.0], "gdp_pdf_aa": [0.0, 100.0], "gdpval_aa": [0.0, 100.0], "gpqa_diamond": [20.0, 80.0], "graphwalks_bfs_256k_1m": [0.0, 85.0], "graphwalks_parents_256k_1m": [0.0, 100.0], "gsm8k": [20.0, 100.0], "healthbench_hard": [0.0, 50.0], "hellaswag": [40.0, 100.0], "helm_safety": [30.0, 100.0], "hle": [0.0, 65.0], "hle_tools": [0.0, 70.0], "humaneval": [10.0, 100.0], "humaneval_plus": [10.0, 100.0], "ifeval": [20.0, 95.0], "ipho_2025_theory": [0.0, 100.0], "lab_bench_figqa": [20.0, 90.0], "lab_bench_figqa_tools": [20.0, 90.0], "legalbench": [0.0, 100.0], "live_code_bench": [0.0, 60.0], "math_500": [10.0, 100.0], "mathvista": [20.0, 80.0], "mbpp": [20.0, 100.0], "medmcqa": [20.0, 80.0], "medqa": [0.0, 100.0], "medxpertqa_multimodal": [20.0, 85.0], "mgsm": [10.0, 100.0], "miracl": [10.0, 70.0], "mmlu_abstract_algebra": [20.0, 75.0], "mmlu_anatomy": [20.0, 90.0], "mmlu_astronomy": [20.0, 95.0], "mmlu_biology": [20.0, 95.0], "mmlu_business_ethics": [20.0, 90.0], "mmlu_chemistry": [20.0, 95.0], "mmlu_clinical_knowledge": [20.0, 95.0], "mmlu_college_biology": [20.0, 95.0], "mmlu_college_chemistry": [20.0, 80.0], "mmlu_college_computer_science": [20.0, 95.0], "mmlu_college_mathematics": [20.0, 80.0], "mmlu_college_medicine": [20.0, 90.0], "mmlu_college_physics": [20.0, 80.0], "mmlu_computer_science": [20.0, 95.0], "mmlu_computer_security": [20.0, 95.0], "mmlu_conceptual_physics": [20.0, 95.0], "mmlu_econometrics": [20.0, 85.0], "mmlu_electrical_engineering": [20.0, 90.0], "mmlu_elementary_mathematics": [20.0, 90.0], "mmlu_formal_logic": [20.0, 80.0], "mmlu_global_facts": [20.0, 75.0], "mmlu_high_school_biology": [20.0, 95.0], "mmlu_high_school_chemistry": [20.0, 90.0], "mmlu_high_school_computer_science": [20.0, 95.0], "mmlu_high_school_european_history": [20.0, 95.0], "mmlu_high_school_geography": [20.0, 95.0], "mmlu_high_school_government_and_politics": [20.0, 100.0], "mmlu_high_school_macroeconomics": [20.0, 95.0], "mmlu_high_school_mathematics": [20.0, 80.0], "mmlu_high_school_microeconomics": [20.0, 95.0], "mmlu_high_school_physics": [20.0, 85.0], "mmlu_high_school_psychology": [20.0, 98.0], "mmlu_high_school_statistics": [20.0, 90.0], "mmlu_high_school_us_history": [20.0, 95.0], "mmlu_high_school_world_history": [20.0, 95.0], "mmlu_human_aging": [20.0, 90.0], "mmlu_human_sexuality": [20.0, 95.0], "mmlu_international_law": [20.0, 95.0], "mmlu_jurisprudence": [20.0, 90.0], "mmlu_logical_fallacies": [20.0, 95.0], "mmlu_machine_learning": [20.0, 85.0], "mmlu_management": [20.0, 95.0], "mmlu_marketing": [20.0, 95.0], "mmlu_medical_genetics": [20.0, 95.0], "mmlu_miscellaneous": [20.0, 95.0], "mmlu_moral_disputes": [20.0, 90.0], "mmlu_moral_scenarios": [20.0, 85.0], "mmlu_nutrition": [20.0, 95.0], "mmlu_philosophy": [20.0, 90.0], "mmlu_physics": [20.0, 95.0], "mmlu_prehistory": [20.0, 95.0], "mmlu_pro": [20.0, 90.0], "mmlu_professional_accounting": [20.0, 85.0], "mmlu_professional_law": [20.0, 90.0], "mmlu_professional_medicine": [20.0, 95.0], "mmlu_professional_psychology": [20.0, 95.0], "mmlu_public_relations": [20.0, 90.0], "mmlu_security_studies": [20.0, 90.0], "mmlu_sociology": [20.0, 95.0], "mmlu_us_foreign_policy": [20.0, 95.0], "mmlu_virology": [20.0, 80.0], "mmlu_world_religions": [20.0, 95.0], "mmmlu": [40.0, 95.0], "mmmu": [20.0, 80.0], "mos_tts": [1.0, 5.0], "mt_bench": [5.0, 10.0], "mteb_classification": [40.0, 90.0], "mteb_clustering": [20.0, 60.0], "mteb_overall": [30.0, 80.0], "mteb_pair_classification": [50.0, 95.0], "mteb_reranking": [20.0, 70.0], "mteb_retrieval": [20.0, 70.0], "mteb_sts": [40.0, 90.0], "mteb_summarization": [20.0, 50.0], "multipl_e": [10.0, 100.0], "multipl_e_cpp": [10.0, 100.0], "multipl_e_csharp": [10.0, 100.0], "multipl_e_go": [10.0, 100.0], "multipl_e_java": [10.0, 100.0], "multipl_e_javascript": [10.0, 100.0], "multipl_e_julia": [10.0, 100.0], "multipl_e_kotlin": [10.0, 100.0], "multipl_e_lua": [10.0, 100.0], "multipl_e_perl": [10.0, 100.0], "multipl_e_php": [10.0, 100.0], "multipl_e_python": [10.0, 100.0], "multipl_e_r": [10.0, 100.0], "multipl_e_ruby": [10.0, 100.0], "multipl_e_rust": [10.0, 100.0], "multipl_e_scala": [10.0, 100.0], "multipl_e_swift": [10.0, 100.0], "multipl_e_typescript": [10.0, 100.0], "musr": [10.0, 80.0], "ocrbench": [0.0, 100.0], "osworld": [0.0, 80.0], "pubmedqa": [30.0, 90.0], "realworldqa": [30.0, 90.0], "scicode": [0.0, 100.0], "screenspot_pro": [10.0, 95.0], "screenspot_pro_tools": [10.0, 95.0], "swe_bench_agent": [0.0, 60.0], "swe_bench_multilingual": [0.0, 90.0], "swe_bench_multimodal": [0.0, 60.0], "swe_bench_pro": [0.0, 80.0], "swe_bench_verified": [0.0, 70.0], "tau_bench": [0.0, 80.0], "terminal_bench": [0.0, 80.0], "terminal_bench_2": [0.0, 80.0], "toxigen": [30.0, 100.0], "truthfulqa": [20.0, 90.0], "usamo_2026": [0.0, 100.0], "web_arena": [0.0, 50.0], "wer_librispeech": [1.5, 25.0], "wildbench": [-100.0, 100.0], "winogrande": [50.0, 100.0], "zerobench": [0.0, 50.0]}, "build": {"built_at": "2026-09-19T13:10:56+00:00", "commit": "22281ce9000688279886f11927c34606b427b4b7", "eligibility_as_of": "2026-09-09", "export_schema_version": "2.0"}, "featured": ["general", "coding", "reasoning", "chat", "agentic", "rag", "vision", "multilingual", "math_competition", "writing_technical", "summarization", "embedding", "text_to_speech"], "profiles": {"accounting": {"benchmark_weights": {"arena_elo_overall": 0.15, "finbench": 0.2, "finqa": 0.15, "ifeval": 0.1, "math_500": 0.1, "mmlu_professional_accounting": 0.3}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.25}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "agentic": {"benchmark_weights": {"aa_briefcase": 0.0667, "arena_elo_overall": 0.12, "automationbench_aa": 0.0667, "gdpval_aa": 0.0667, "ifeval": 0.08, "swe_bench_agent": 0.16, "swe_bench_verified": 0.12, "tau_bench": 0.12, "terminal_bench": 0.08, "web_arena": 0.12}, "capability_weights": {"coding": 0.2, "reasoning": 0.2, "tool_use": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-code", "llm-chat"]}, "biotech": {"benchmark_weights": {"arena_elo_overall": 0.1, "gpqa_diamond": 0.15, "medqa": 0.15, "mmlu_biology": 0.2, "mmlu_chemistry": 0.15, "mmlu_pro": 0.1, "pubmedqa": 0.15}, "capability_weights": {"domain": 0.2, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "chat": {"benchmark_weights": {"alpaca_eval": 0.15, "arena_elo_overall": 0.3, "arena_elo_style_control": 0.1, "ifeval": 0.15, "mmlu_pro": 0.1, "mt_bench": 0.15, "wildbench": 0.05}, "capability_weights": {"creative": 0.2, "language": 0.2, "reasoning": 0.15, "tool_use": 0.1}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-chat", "vlm", "llm-reasoning"]}, "code_review": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.15, "arena_elo_overall": 0.1, "humaneval": 0.15, "ifeval": 0.1, "swe_bench_verified": 0.25, "terminal_bench": 0.1}, "capability_weights": {"coding": 0.3, "language": 0.1, "reasoning": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-reasoning", "llm-chat"]}, "coding": {"benchmark_weights": {"aider_polyglot": 0.12, "arena_elo_coding": 0.12, "arena_elo_overall": 0.08, "humaneval": 0.16, "live_code_bench": 0.12, "scicode": 0.2, "swe_bench_verified": 0.16, "terminal_bench": 0.04}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "coding_cpp": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.1, "arena_elo_overall": 0.05, "humaneval": 0.15, "multipl_e": 0.1, "multipl_e_cpp": 0.3, "swe_bench_verified": 0.15}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "coding_go": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.1, "arena_elo_overall": 0.05, "humaneval": 0.15, "multipl_e": 0.1, "multipl_e_go": 0.3, "swe_bench_verified": 0.15}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "coding_java": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.1, "arena_elo_overall": 0.05, "humaneval": 0.15, "multipl_e": 0.1, "multipl_e_java": 0.3, "swe_bench_verified": 0.15}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "coding_javascript": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.1, "arena_elo_overall": 0.05, "humaneval": 0.15, "multipl_e": 0.1, "multipl_e_javascript": 0.3, "swe_bench_verified": 0.15}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "coding_python": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.1, "arena_elo_overall": 0.05, "humaneval": 0.15, "multipl_e": 0.1, "multipl_e_python": 0.3, "swe_bench_verified": 0.15}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "coding_rust": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.1, "arena_elo_overall": 0.05, "humaneval": 0.15, "multipl_e": 0.1, "multipl_e_rust": 0.3, "swe_bench_verified": 0.15}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "coding_typescript": {"benchmark_weights": {"aider_polyglot": 0.15, "arena_elo_coding": 0.1, "arena_elo_overall": 0.05, "humaneval": 0.15, "multipl_e": 0.1, "multipl_e_typescript": 0.3, "swe_bench_verified": 0.15}, "capability_weights": {"coding": 0.3, "reasoning": 0.2, "tool_use": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"]}, "content_moderation": {"benchmark_weights": {"arena_elo_overall": 0.15, "bbq": 0.2, "helm_safety": 0.3, "ifeval": 0.1, "toxigen": 0.25}, "capability_weights": {"language": 0.15, "safety": 0.3}, "context_weight": 0.05, "cost_weight": 0.0, "preferred_types": ["safety-classifier", "llm-chat", "reward-model"]}, "customer_support": {"benchmark_weights": {"alpaca_eval": 0.15, "arena_elo_overall": 0.2, "arena_elo_style_control": 0.15, "ifeval": 0.2, "mt_bench": 0.2, "truthfulqa": 0.1}, "capability_weights": {"creative": 0.1, "language": 0.25, "tool_use": 0.2}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-chat", "vlm"]}, "cybersecurity": {"benchmark_weights": {"arena_elo_coding": 0.1, "arena_elo_overall": 0.1, "gpqa_diamond": 0.1, "humaneval": 0.15, "ifeval": 0.1, "mmlu_computer_science": 0.15, "swe_bench_verified": 0.15, "terminal_bench": 0.15}, "capability_weights": {"coding": 0.25, "reasoning": 0.25, "tool_use": 0.2}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-reasoning", "llm-chat"]}, "data_science": {"benchmark_weights": {"aider_polyglot": 0.1, "arena_elo_coding": 0.15, "arena_elo_overall": 0.05, "gpqa_diamond": 0.1, "humaneval": 0.15, "math_500": 0.15, "mmlu_pro": 0.1, "multipl_e_python": 0.2}, "capability_weights": {"coding": 0.25, "reasoning": 0.25, "tool_use": 0.15}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-reasoning", "llm-chat"]}, "devops": {"benchmark_weights": {"aider_polyglot": 0.1, "arena_elo_coding": 0.15, "arena_elo_overall": 0.1, "humaneval": 0.15, "ifeval": 0.1, "swe_bench_verified": 0.15, "terminal_bench": 0.25}, "capability_weights": {"coding": 0.25, "reasoning": 0.15, "tool_use": 0.25}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-code", "llm-chat", "llm-reasoning"]}, "education": {"benchmark_weights": {"arc_challenge": 0.15, "arena_elo_overall": 0.15, "hellaswag": 0.1, "ifeval": 0.1, "mmlu_pro": 0.25, "mt_bench": 0.15, "truthfulqa": 0.1}, "capability_weights": {"creative": 0.15, "language": 0.2, "reasoning": 0.2}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning", "vlm"]}, "education_humanities": {"benchmark_weights": {"alpaca_eval": 0.15, "arena_elo_overall": 0.2, "mmlu_business_ethics": 0.1, "mmlu_pro": 0.15, "mmlu_professional_law": 0.1, "mt_bench": 0.15, "truthfulqa": 0.15}, "capability_weights": {"creative": 0.2, "language": 0.25, "reasoning": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "education_stem": {"benchmark_weights": {"arena_elo_overall": 0.15, "gpqa_diamond": 0.15, "math_500": 0.15, "mmlu_biology": 0.1, "mmlu_chemistry": 0.15, "mmlu_physics": 0.15, "mmlu_pro": 0.15}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "embedding": {"benchmark_weights": {"beir": 0.15, "miracl": 0.1, "mteb_classification": 0.15, "mteb_clustering": 0.05, "mteb_overall": 0.3, "mteb_retrieval": 0.25}, "capability_weights": {}, "context_weight": 0.05, "cost_weight": 0.0, "preferred_types": ["embedding-text", "embedding-multimodal"]}, "financial": {"benchmark_weights": {"arena_elo_overall": 0.15, "finbench": 0.3, "finqa": 0.2, "ifeval": 0.1, "math_500": 0.1, "mmlu_pro": 0.15}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "financial_analysis": {"benchmark_weights": {"arena_elo_overall": 0.15, "finbench": 0.25, "finqa": 0.25, "math_500": 0.15, "mmlu_pro": 0.1, "mmlu_professional_accounting": 0.1}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "financial_compliance": {"benchmark_weights": {"arena_elo_overall": 0.15, "finbench": 0.2, "ifeval": 0.15, "legalbench": 0.2, "mmlu_professional_accounting": 0.15, "mmlu_professional_law": 0.15}, "capability_weights": {"domain": 0.15, "language": 0.15, "reasoning": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "general": {"benchmark_weights": {"arena_elo_overall": 0.2, "gdpval_aa": 0.2, "gpqa_diamond": 0.08, "humaneval": 0.08, "ifeval": 0.08, "math_500": 0.08, "mmlu_pro": 0.12, "mt_bench": 0.08, "swe_bench_verified": 0.08}, "capability_weights": {"coding": 0.15, "creative": 0.1, "reasoning": 0.15, "tool_use": 0.1}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning", "vlm"]}, "image_generation": {"benchmark_weights": {"arena_elo_overall": 0.2, "arena_elo_vision": 0.2, "clip_score": 0.3, "fid": 0.3}, "capability_weights": {"creative": 0.3}, "context_weight": 0.05, "cost_weight": 0.0, "preferred_types": ["image-gen", "vlm"]}, "legal": {"benchmark_weights": {"arena_elo_overall": 0.1, "ifeval": 0.1, "legalbench": 0.35, "mmlu_jurisprudence": 0.15, "mmlu_pro": 0.1, "mmlu_professional_law": 0.2}, "capability_weights": {"domain": 0.15, "language": 0.15, "reasoning": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "legal_contract_review": {"benchmark_weights": {"arena_elo_overall": 0.1, "ifeval": 0.15, "legalbench": 0.3, "mmlu_pro": 0.15, "mmlu_professional_law": 0.2, "mt_bench": 0.1}, "capability_weights": {"domain": 0.15, "language": 0.2, "reasoning": 0.25}, "context_weight": 0.2, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "math_competition": {"benchmark_weights": {"aime_2025": 0.3, "aime_2026": 0.15, "arena_elo_math": 0.1, "gpqa_diamond": 0.1, "gsm8k": 0.1, "math_500": 0.25}, "capability_weights": {"coding": 0.1, "reasoning": 0.35}, "context_weight": 0.05, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "medical": {"benchmark_weights": {"arena_elo_overall": 0.1, "medmcqa": 0.15, "medqa": 0.3, "mmlu_clinical_knowledge": 0.15, "mmlu_pro": 0.1, "pubmedqa": 0.2}, "capability_weights": {"domain": 0.2, "language": 0.1, "reasoning": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "medical_clinical": {"benchmark_weights": {"arena_elo_overall": 0.1, "medmcqa": 0.15, "medqa": 0.25, "mmlu_clinical_knowledge": 0.25, "mmlu_pro": 0.1, "pubmedqa": 0.15}, "capability_weights": {"domain": 0.2, "language": 0.1, "reasoning": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "medical_radiology": {"benchmark_weights": {"arena_elo_overall": 0.15, "arena_elo_vision": 0.15, "medqa": 0.2, "mmlu_clinical_knowledge": 0.15, "mmmu": 0.2, "pubmedqa": 0.15}, "capability_weights": {"creative": 0.1, "domain": 0.15, "reasoning": 0.25}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["vlm", "llm-reasoning", "llm-chat"]}, "multilingual": {"benchmark_weights": {"arena_elo_overall": 0.15, "flores": 0.2, "ifeval": 0.1, "mgsm": 0.25, "miracl": 0.2, "mmlu_pro": 0.1}, "capability_weights": {"creative": 0.1, "language": 0.35, "reasoning": 0.1}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "vlm"]}, "rag": {"benchmark_weights": {"aa_lcr": 0.1, "arena_elo_overall": 0.12, "beir": 0.16, "gdp_pdf_aa": 0.1, "ifeval": 0.08, "miracl": 0.04, "mmlu_pro": 0.08, "mteb_overall": 0.12, "mteb_retrieval": 0.2}, "capability_weights": {"language": 0.15, "reasoning": 0.1}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["embedding-text", "reranker", "llm-chat"]}, "reasoning": {"benchmark_weights": {"aime_2025": 0.12, "arena_elo_overall": 0.12, "bbh": 0.08, "critpt": 0.2, "gpqa_diamond": 0.16, "ifeval": 0.04, "math_500": 0.16, "mmlu_pro": 0.12}, "capability_weights": {"coding": 0.15, "reasoning": 0.35, "tool_use": 0.1}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "research_assistant": {"benchmark_weights": {"arena_elo_hard_prompts": 0.1, "arena_elo_overall": 0.15, "gpqa_diamond": 0.2, "ifeval": 0.1, "math_500": 0.1, "mmlu_pro": 0.15, "mt_bench": 0.1, "truthfulqa": 0.1}, "capability_weights": {"language": 0.15, "reasoning": 0.25, "tool_use": 0.15}, "context_weight": 0.2, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat", "vlm"]}, "roleplay": {"benchmark_weights": {"alpaca_eval": 0.2, "arena_elo_overall": 0.15, "arena_elo_style_control": 0.25, "ifeval": 0.1, "mt_bench": 0.15, "wildbench": 0.15}, "capability_weights": {"creative": 0.35, "language": 0.25}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat"]}, "safety": {"benchmark_weights": {"arena_elo_overall": 0.2, "bbq": 0.25, "helm_safety": 0.3, "toxigen": 0.25}, "capability_weights": {}, "context_weight": 0.05, "cost_weight": 0.0, "preferred_types": ["safety-classifier", "reward-model"]}, "science": {"benchmark_weights": {"arena_elo_overall": 0.08, "critpt": 0.1, "gpqa_diamond": 0.2, "math_500": 0.12, "mmlu_biology": 0.08, "mmlu_chemistry": 0.08, "mmlu_physics": 0.08, "mmlu_pro": 0.16, "scicode": 0.1}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "science_astronomy": {"benchmark_weights": {"arena_elo_overall": 0.1, "gpqa_diamond": 0.2, "math_500": 0.15, "mmlu_astronomy": 0.3, "mmlu_physics": 0.15, "mmlu_pro": 0.1}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "science_biology": {"benchmark_weights": {"arena_elo_overall": 0.15, "gpqa_diamond": 0.2, "ifeval": 0.1, "mmlu_biology": 0.3, "mmlu_clinical_knowledge": 0.1, "mmlu_pro": 0.15}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "science_chemistry": {"benchmark_weights": {"arena_elo_overall": 0.1, "gpqa_diamond": 0.2, "ifeval": 0.1, "math_500": 0.15, "mmlu_chemistry": 0.3, "mmlu_pro": 0.15}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "science_physics": {"benchmark_weights": {"arena_elo_overall": 0.1, "gpqa_diamond": 0.2, "ifeval": 0.1, "math_500": 0.15, "mmlu_physics": 0.3, "mmlu_pro": 0.15}, "capability_weights": {"domain": 0.15, "language": 0.1, "reasoning": 0.3}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-reasoning", "llm-chat"]}, "speech_to_text": {"benchmark_weights": {"arena_elo_overall": 0.2, "mgsm": 0.15, "miracl": 0.15, "wer_librispeech": 0.5}, "capability_weights": {"language": 0.2}, "context_weight": 0.05, "cost_weight": 0.0, "preferred_types": ["audio-stt", "audio-multimodal"]}, "summarization": {"benchmark_weights": {"alpaca_eval": 0.2, "arena_elo_overall": 0.15, "arena_elo_style_control": 0.15, "ifeval": 0.15, "mt_bench": 0.2, "wildbench": 0.15}, "capability_weights": {"creative": 0.25, "language": 0.2, "reasoning": 0.15}, "context_weight": 0.2, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "text_to_speech": {"benchmark_weights": {"arena_elo_overall": 0.2, "arena_elo_style_control": 0.15, "mos_tts": 0.5, "mt_bench": 0.15}, "capability_weights": {"creative": 0.15, "language": 0.15}, "context_weight": 0.05, "cost_weight": 0.0, "preferred_types": ["audio-tts", "audio-multimodal"]}, "translation": {"benchmark_weights": {"arena_elo_overall": 0.1, "flores": 0.3, "ifeval": 0.1, "mgsm": 0.2, "miracl": 0.2, "mmlu_pro": 0.1}, "capability_weights": {"creative": 0.15, "language": 0.3, "reasoning": 0.1}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "vlm"]}, "vision": {"benchmark_weights": {"arena_elo_overall": 0.1, "arena_elo_vision": 0.15, "chartqa": 0.15, "docvqa": 0.15, "mathvista": 0.2, "mmmu": 0.25}, "capability_weights": {"creative": 0.1, "reasoning": 0.15}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["vlm", "llm-chat"]}, "writing_creative": {"benchmark_weights": {"alpaca_eval": 0.25, "arena_elo_overall": 0.1, "arena_elo_style_control": 0.15, "ifeval": 0.1, "mt_bench": 0.2, "wildbench": 0.2}, "capability_weights": {"creative": 0.3, "language": 0.2, "reasoning": 0.1}, "context_weight": 0.1, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning"]}, "writing_technical": {"benchmark_weights": {"alpaca_eval": 0.15, "arena_elo_overall": 0.15, "ifeval": 0.2, "mmlu_pro": 0.15, "mt_bench": 0.2, "wildbench": 0.15}, "capability_weights": {"creative": 0.25, "language": 0.15, "reasoning": 0.2}, "context_weight": 0.15, "cost_weight": 0.0, "preferred_types": ["llm-chat", "llm-reasoning", "llm-code"]}}, "ranking_policy": {"cli_min_benchmark_coverage": 0.5, "limit_applies_to": "ranked_only", "min_benchmark_count": 2, "min_benchmark_coverage": 0.5, "neutrality": {"assertions": {"accepts_paid_placement": false, "accepts_provider_paid_visibility": false, "accepts_referral_fees": false, "proxies_inference_tokens": false, "stores_customer_prompts": false}, "charges": "the consumer of a recommendation, never its subjects", "method_source": "https://github.com/turbobeest/modelspec/blob/main/api/ranking/engine.py", "neutrality_url": "https://modelspec.dev/legal/neutrality/", "operator": "Sparks & Sawdust LLC", "permanent": true, "pledge": "No referral fees, no paid placement, no provider-paid visibility, permanently.", "privacy_url": "https://modelspec.dev/legal/privacy/", "rule": "Charging the consumer of a recommendation is compatible with being an honest broker. Charging the subjects of one is not.", "source_neutral_at": ["ranking", "tie_breaks", "hosting_suggestions", "route_advice"], "terms_url": "https://modelspec.dev/legal/terms/", "version": "neutrality-v1"}, "ordering": "conservative_lower_bound", "uncertainty": "missing-benchmark bounds, not statistical confidence intervals", "version": "incomplete-evidence-v1", "wizard_min_benchmark_coverage": 0.25}}