{"generated_at":"2026-09-28T07:29:13.606Z","count":74,"result_count":431,"benchmarks":[{"id":"swe-bench-verified","name":"SWE-bench Verified","domain":"coding","subdomain":"repo-level-bugfix","url":"https://www.swebench.com/","source":{"type":"leaderboard_api","adapter":"swebench"},"what_it_measures":"Whether a model fixes a real bug in a real Python repository such that the project's own tests pass. Measures agentic work on existing code.","what_it_measures_de":"Ob ein Modell einen echten Fehler in einem echten Python-Repository so behebt, dass die projekteigenen Tests durchlaufen. Misst agentische Arbeit an bestehendem Code, nicht das Schreiben von Schnipseln.","what_it_does_not_measure":"Languages other than Python, frontend work, architectural decisions, readability.","what_it_does_not_measure_de":"Andere Sprachen als Python, Frontend-Arbeit, Architekturentscheidungen, Lesbarkeit.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"sample_size":500,"contamination_risk":"medium","difficulty":"high","answer_format":"patch","marketing_weight":"high","open_harness":true,"scaffold_documented":false,"notes_de":"Die Aufgaben stammen aus oeffentlichen GitHub-Issues, Kontamination ist also plausibel. Herstellerzahlen schwanken stark mit dem Agent-Geruest; nur Werte mit dokumentiertem Scaffold sind vergleichbar.","benchmaxxing_risk":"medium","axes":{"recency":0.93,"discrimination":0.59,"headroom":0.74,"contamination":0.55,"benchmaxxing":0.55,"rigor":1,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z","top_score":77.8,"top5_spread":7.1,"sample_results":6,"last_result_update":"2026-09-01","source_last_update":"2026-09-01"},"evidence_score":71,"status":"active","result_count":6,"url_local":"https://modelradar-one.vercel.app/benchmarks/swe-bench-verified"},{"id":"swe-bench-multilingual","name":"SWE-bench Multilingual","domain":"coding","subdomain":"repo-level-bugfix","url":"https://www.swebench.com/multilingual.html","source":{"type":"manual"},"what_it_measures":"Repo-level bug fixing across nine programming languages.","what_it_measures_de":"Fehlerbehebung auf Repository-Ebene in neun Programmiersprachen.","what_it_does_not_measure_de":"Neuentwicklung, Architektur, Performance-Optimierung.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"sample_size":300,"contamination_risk":"medium","difficulty":"high","open_harness":true,"notes_de":"Deckt die Python-Schlagseite von SWE-bench Verified ab.","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":58,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/swe-bench-multilingual"},{"id":"swe-bench-pro","name":"SWE-bench Pro","domain":"coding","subdomain":"repo-level-bugfix","url":"https://www.swebench.com/","source":{"type":"manual"},"what_it_measures_de":"Schwierigere, laengere Aufgaben als SWE-bench Verified, teils mit privatem Testsatz gegen Kontamination.","what_it_does_not_measure_de":"Kurze Einzeldatei-Aenderungen.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":false,"private_holdout":true,"human_verified":true,"llm_judged":false,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":69,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/swe-bench-pro"},{"id":"aider-polyglot","name":"Aider Polyglot","domain":"coding","subdomain":"edit-format","url":"https://aider.chat/docs/leaderboards/","source":{"type":"leaderboard_api","adapter":"aider"},"what_it_measures_de":"Ob ein Modell Codeaenderungen in mehreren Sprachen korrekt UND im geforderten Bearbeitungsformat ausliefert. Das Format ist der eigentliche Pruefstein.","what_it_does_not_measure_de":"Agentische Mehrschritt-Arbeit, Werkzeugnutzung.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":false,"llm_judged":false,"sample_size":225,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"scaffold_documented":true,"notes_de":"Praxisnah, weil Formattreue im Alltag oft mehr kostet als Denkleistung.","benchmaxxing_risk":"medium","axes":{"recency":0.01,"discrimination":0.5,"headroom":0.5,"contamination":0.55,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":1},"computed_at":"2026-09-28T04:00:16.321Z","source_last_update":"2025-10-03","last_result_update":"2025-10-03"},"evidence_score":42,"status":"stale","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/aider-polyglot"},{"id":"livecodebench","name":"LiveCodeBench","domain":"coding","subdomain":"competitive-programming","url":"https://livecodebench.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Wettbewerbsaufgaben, die NACH dem Trainingsschnitt eines Modells veroeffentlicht wurden. Genau dieser Zeitschnitt ist der Sinn der Sache.","what_it_does_not_measure_de":"Arbeit an bestehendem Code, Lesbarkeit, Wartbarkeit.","scale":{"min":0,"max":100,"unit":"% pass@1","higher_is_better":true},"rating":{"test_set_public":true,"private_holdout":true,"human_verified":false,"llm_judged":false,"contamination_risk":"low","difficulty":"high","open_harness":true,"notes_de":"Kontaminationsarm durch das rollierende Zeitfenster. Nur mit angegebenem Zeitraum vergleichbar, sonst misst man verschiedene Aufgabensaetze.","benchmaxxing_risk":"low","axes":{"recency":0.81,"discrimination":1,"headroom":0.37,"contamination":1,"benchmaxxing":1,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z","top_score":89,"top5_spread":32.5,"sample_results":4,"last_result_update":"2026-07-22"},"evidence_score":80,"status":"active","result_count":4,"url_local":"https://modelradar-one.vercel.app/benchmarks/livecodebench"},{"id":"humaneval","name":"HumanEval","domain":"coding","subdomain":"function-synthesis","url":"https://github.com/openai/human-eval","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"164 kleine Python-Funktionen aus einer Docstring-Beschreibung.","what_it_does_not_measure_de":"Alles, was ueber eine einzelne kurze Funktion hinausgeht.","scale":{"min":0,"max":100,"unit":"% pass@1","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"sample_size":164,"contamination_risk":"high","difficulty":"low","answer_format":"code","marketing_weight":"high","open_harness":true,"notes_de":"Historisch wichtig, heute praktisch ausgereizt und stark kontaminiert. Wird hier gefuehrt, damit alte Herstellerangaben einordbar bleiben, nicht weil er noch etwas unterscheidet.","benchmaxxing_risk":"medium","axes":{"recency":0.95,"discrimination":1,"headroom":0.3,"contamination":0.15,"benchmaxxing":0.55,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z","top_score":90.9,"top5_spread":28.1,"sample_results":5,"last_result_update":"2026-09-10"},"evidence_score":57,"status":"saturated","result_count":5,"url_local":"https://modelradar-one.vercel.app/benchmarks/humaneval"},{"id":"bigcodebench","name":"BigCodeBench","domain":"coding","subdomain":"library-use","url":"https://bigcode-bench.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Aufgaben, die den korrekten Einsatz echter Bibliotheken verlangen, nicht nur Algorithmik.","what_it_does_not_measure_de":"Mehrdateien-Aenderungen, Projektkontext.","scale":{"min":0,"max":100,"unit":"% pass@1","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"sample_size":1140,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.95,"discrimination":0,"headroom":1,"contamination":0.55,"benchmaxxing":1,"rigor":1,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z","top_score":56.8,"top5_spread":0,"sample_results":2,"last_result_update":"2026-09-10"},"evidence_score":66,"status":"active","result_count":2,"url_local":"https://modelradar-one.vercel.app/benchmarks/bigcodebench"},{"id":"terminal-bench","name":"Terminal-Bench","domain":"coding","subdomain":"shell-agent","url":"https://www.tbench.ai/","source":{"type":"manual"},"what_it_measures_de":"Ob ein Agent Aufgaben in einer echten Terminalumgebung zu Ende bringt — Installieren, Bauen, Debuggen, Aufraeumen.","what_it_does_not_measure_de":"Reine Codequalitaet, Sprachvielfalt.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"contamination_risk":"low","difficulty":"high","open_harness":true,"scaffold_documented":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":1,"rigor":0.8,"reproducibility":1},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":65,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/terminal-bench"},{"id":"crux-eval","name":"CRUXEval","domain":"coding","subdomain":"code-reasoning","url":"https://crux-eval.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Ob ein Modell Ein- und Ausgabe von Code vorhersagen kann, ohne ihn auszufuehren.","what_it_does_not_measure_de":"Codeerzeugung.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"llm_judged":false,"sample_size":800,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.7,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":48,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/crux-eval"},{"id":"repobench","name":"RepoBench","domain":"coding","subdomain":"repo-completion","url":"https://github.com/Leolty/repobench","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Codevervollstaendigung mit Kontext aus dem umgebenden Repository.","what_it_does_not_measure_de":"Agentisches Vorgehen, Testausfuehrung.","scale":{"min":0,"max":100,"unit":"% exact/edit","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":46,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/repobench"},{"id":"mbpp-plus","name":"MBPP+","domain":"coding","subdomain":"function-synthesis","url":"https://evalplus.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"MBPP mit deutlich strengeren Testfaellen — deckt Scheinlosungen auf.","what_it_does_not_measure_de":"Projektarbeit, Bibliotheksnutzung.","scale":{"min":0,"max":100,"unit":"% pass@1","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"low","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":40,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/mbpp-plus"},{"id":"swe-lancer","name":"SWE-Lancer","domain":"coding","subdomain":"economic-value","url":"https://openai.com/index/swe-lancer/","source":{"type":"manual"},"what_it_measures_de":"Echte bezahlte Freelance-Aufgaben; bewertet wird in Dollar geloester Auftragswert.","what_it_does_not_measure_de":"Grundlagenwissen, Geschwindigkeit.","scale":{"min":0,"max":1000000,"unit":"USD geloest","higher_is_better":true},"rating":{"test_set_public":false,"human_verified":true,"contamination_risk":"low","difficulty":"high","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":67,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/swe-lancer"},{"id":"tau-bench","name":"TAU-bench","domain":"agentic","subdomain":"customer-workflows","url":"https://github.com/sierra-research/tau-bench","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Ob ein Agent mehrstufige Kundenvorgaenge korrekt abschliesst und dabei Regeln einhaelt, statt nur plausibel zu antworten.","what_it_does_not_measure_de":"Codearbeit, Wissen, Kreativitaet.","scale":{"min":0,"max":100,"unit":"% pass^1","higher_is_better":true},"rating":{"test_set_public":true,"llm_judged":false,"contamination_risk":"low","difficulty":"high","open_harness":true,"scaffold_documented":true,"notes_de":"Misst Regeltreue unter Druck, was in der Praxis oft wichtiger ist als Rohleistung.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":1},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":55,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/tau-bench"},{"id":"bfcl","name":"Berkeley Function Calling Leaderboard","domain":"tool-use","subdomain":"function-calling","url":"https://gorilla.cs.berkeley.edu/leaderboard.html","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Ob Funktionsaufrufe syntaktisch und semantisch korrekt sind, inklusive Mehrfach- und Parallelaufrufen sowie dem Erkennen, wann NICHT aufzurufen ist.","what_it_does_not_measure_de":"Laengere Agentenketten, Weltwissen.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":55,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/bfcl"},{"id":"gaia","name":"GAIA","domain":"agentic","subdomain":"general-assistant","url":"https://huggingface.co/spaces/gaia-benchmark/leaderboard","source":{"type":"manual"},"what_it_measures_de":"Alltagsaufgaben, die Werkzeuge, Websuche und mehrere Schritte brauchen — fuer Menschen leicht, fuer Modelle lange schwer.","what_it_does_not_measure_de":"Fachwissen in der Tiefe, Codequalitaet.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":false,"private_holdout":true,"human_verified":true,"llm_judged":false,"sample_size":466,"contamination_risk":"low","difficulty":"high","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":67,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/gaia"},{"id":"osworld","name":"OSWorld","domain":"agentic","subdomain":"computer-use","url":"https://os-world.github.io/","source":{"type":"manual"},"what_it_measures_de":"Ob ein Agent Aufgaben in einer echten Desktop-Umgebung erledigt.","what_it_does_not_measure_de":"Textqualitaet, Wissen.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":63,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/osworld"},{"id":"webarena","name":"WebArena","domain":"agentic","subdomain":"web-navigation","url":"https://webarena.dev/","source":{"type":"manual"},"what_it_measures_de":"Aufgaben in nachgebauten, funktionierenden Webanwendungen.","what_it_does_not_measure_de":"Offenes Web mit Werbung, Logins und Bruechen.","scale":{"min":0,"max":100,"unit":"% erfolgreich","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"medium","difficulty":"high","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.321Z"},"evidence_score":49,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/webarena"},{"id":"browsecomp","name":"BrowseComp","domain":"agentic","subdomain":"web-research","url":"https://openai.com/index/browsecomp/","source":{"type":"manual"},"what_it_measures_de":"Ob ein Agent schwer auffindbare Fakten im offenen Web zusammensucht. Die Antworten sind kurz und eindeutig pruefbar.","what_it_does_not_measure_de":"Schreibqualitaet, Zusammenfassen.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":false,"private_holdout":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":67,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/browsecomp"},{"id":"mle-bench","name":"MLE-bench","domain":"agentic","subdomain":"ml-engineering","url":"https://github.com/openai/mle-bench","source":{"type":"manual"},"what_it_measures_de":"Ob ein Agent Kaggle-Wettbewerbe eigenstaendig bearbeitet.","what_it_does_not_measure_de":"Interaktion, Erklaerbarkeit.","scale":{"min":0,"max":100,"unit":"% Medaillen","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"high","open_harness":true,"notes_de":"Kaggle-Loesungen stehen oeffentlich im Netz — Kontamination ist hier strukturell.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":45,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/mle-bench"},{"id":"agentharm","name":"AgentHarm","domain":"safety","subdomain":"agentic-misuse","url":"https://huggingface.co/datasets/ai-safety-institute/AgentHarm","source":{"type":"manual"},"what_it_measures_de":"Ob ein Agent schaedliche Auftraege ausfuehrt, wenn Werkzeuge bereitstehen.","what_it_does_not_measure_de":"Nuetzlichkeit, Ueberblockierung.","scale":{"min":0,"max":100,"unit":"% Verweigerung","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":53,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/agentharm"},{"id":"gpqa-diamond","name":"GPQA Diamond","domain":"reasoning","subdomain":"expert-science","url":"https://github.com/idavidrein/gpqa","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Naturwissenschaftliche Fragen auf Promotionsniveau, die Fachleute anderer Gebiete auch mit Google nicht loesen.","what_it_does_not_measure_de":"Alltagstauglichkeit, Rechenarbeit, Codearbeit.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"sample_size":198,"contamination_risk":"high","difficulty":"high","answer_format":"multiple_choice","marketing_weight":"high","open_harness":true,"notes_de":"Nur 198 Fragen im Multiple-Choice-Format bei sehr hoher Marketingwirkung. Das ist die klassische Benchmaxxing-Konstellation; Zahlen nahe der Spitze sind mit Vorsicht zu lesen.","benchmaxxing_risk":"high","axes":{"recency":0.95,"discrimination":0.6,"headroom":0.22,"contamination":0.15,"benchmaxxing":0.15,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z","top_score":93.4,"top5_spread":7.2,"sample_results":14,"last_result_update":"2026-09-10"},"evidence_score":50,"status":"saturated","result_count":14,"url_local":"https://modelradar-one.vercel.app/benchmarks/gpqa-diamond"},{"id":"arc-agi-2","name":"ARC-AGI-2","domain":"reasoning","subdomain":"abstract-reasoning","url":"https://arcprize.org/","source":{"type":"manual"},"what_it_measures_de":"Abstraktionsfaehigkeit an Aufgaben, die bewusst nicht im Trainingsmaterial stehen koennen. Der private Testsatz ist der Kern des Verfahrens.","what_it_does_not_measure_de":"Wissen, Sprache, Werkzeugnutzung.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":false,"private_holdout":true,"human_verified":true,"llm_judged":false,"contamination_risk":"low","difficulty":"high","open_harness":true,"scaffold_documented":true,"notes_de":"Einer der wenigen Benchmarks mit echtem, verwaltetem Geheimtestsatz.","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":1},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":71,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/arc-agi-2"},{"id":"hle","name":"Humanity's Last Exam","domain":"reasoning","subdomain":"expert-frontier","url":"https://lastexam.ai/","source":{"type":"manual"},"what_it_measures_de":"Fachfragen am oberen Ende dessen, was Fachleute noch beantworten koennen.","what_it_does_not_measure_de":"Praxisnutzen, Agentenfaehigkeit.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":false,"private_holdout":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","marketing_weight":"high","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":67,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/hle"},{"id":"frontier-math","name":"FrontierMath","domain":"math","subdomain":"research-level","url":"https://epoch.ai/frontiermath","source":{"type":"manual"},"what_it_measures_de":"Ungeloeste bis forschungsnahe Mathematikaufgaben mit eindeutiger Pruefbarkeit.","what_it_does_not_measure_de":"Schulmathematik, Erklaerqualitaet.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":false,"private_holdout":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":67,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/frontier-math"},{"id":"aime","name":"AIME","domain":"math","subdomain":"competition","url":"https://artofproblemsolving.com/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Wettbewerbsmathematik der US-Oberstufe, jaehrlich neue Aufgaben.","what_it_does_not_measure_de":"Beweisfuehrung, angewandte Mathematik.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"sample_size":30,"contamination_risk":"high","difficulty":"high","marketing_weight":"high","notes_de":"Nur 30 Aufgaben je Jahrgang. Ein einzelner Treffer verschiebt das Ergebnis um 3,3 Punkte, was Vergleiche zwischen eng beieinanderliegenden Modellen wertlos macht. Immer den Jahrgang mitlesen.","benchmaxxing_risk":"medium","axes":{"recency":0.94,"discrimination":0.83,"headroom":0.03,"contamination":0.15,"benchmaxxing":0.55,"rigor":0.6,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z","top_score":99.2,"top5_spread":10,"sample_results":10,"last_result_update":"2026-09-08"},"evidence_score":53,"status":"saturated","result_count":10,"url_local":"https://modelradar-one.vercel.app/benchmarks/aime"},{"id":"math-500","name":"MATH-500","domain":"math","subdomain":"school-competition","url":"https://github.com/openai/prm800k","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"500 Aufgaben aus dem MATH-Datensatz, mittlere Schwierigkeit.","what_it_does_not_measure_de":"Forschungsnahe Mathematik.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.84,"discrimination":0,"headroom":0.11,"contamination":0.15,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z","top_score":96.8,"top5_spread":0,"sample_results":1,"last_result_update":"2026-07-31"},"evidence_score":35,"status":"saturated","result_count":1,"url_local":"https://modelradar-one.vercel.app/benchmarks/math-500"},{"id":"hmmt","name":"HMMT","domain":"math","subdomain":"competition","url":"https://www.hmmt.org/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Harvard-MIT-Mathematikwettbewerb, jaehrlich neue Aufgaben.","what_it_does_not_measure_de":"Angewandte Mathematik, Beweise in Textform.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"sample_size":30,"contamination_risk":"medium","difficulty":"high","benchmaxxing_risk":"medium","axes":{"recency":0.94,"discrimination":1,"headroom":0.1,"contamination":0.55,"benchmaxxing":0.55,"rigor":0.3,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z","top_score":96.9,"top5_spread":33.3,"sample_results":6,"last_result_update":"2026-09-08"},"evidence_score":61,"status":"saturated","result_count":6,"url_local":"https://modelradar-one.vercel.app/benchmarks/hmmt"},{"id":"bbh","name":"BIG-Bench Hard","domain":"reasoning","subdomain":"mixed","url":"https://github.com/suzgunmirac/BIG-Bench-Hard","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"23 Teilaufgaben, bei denen Modelle frueher schlechter waren als Menschen.","what_it_does_not_measure_de":"Aktuelle Grenzfaehigkeiten.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"low","open_harness":true,"notes_de":"Von aktuellen Modellen weitgehend ausgereizt.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":40,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/bbh"},{"id":"zebralogic","name":"ZebraLogic","domain":"reasoning","subdomain":"constraint-logic","url":"https://huggingface.co/spaces/allenai/ZebraLogic","source":{"type":"manual"},"what_it_measures_de":"Logikraetsel mit harten Nebenbedingungen, beliebig skalierbare Schwierigkeit.","what_it_does_not_measure_de":"Wissen, Sprache.","scale":{"min":0,"max":100,"unit":"% geloest","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"notes_de":"Aufgaben sind generierbar, daher kaum kontaminierbar.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":53,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/zebralogic"},{"id":"musr","name":"MuSR","domain":"reasoning","subdomain":"multi-step-narrative","url":"https://github.com/Zayne-sprague/MuSR","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Mehrschrittiges Schliessen ueber laengere Erzaehltexte.","what_it_does_not_measure_de":"Fachwissen.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":46,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/musr"},{"id":"mmlu-pro","name":"MMLU-Pro","domain":"knowledge","subdomain":"broad-academic","url":"https://huggingface.co/spaces/TIGER-Lab/MMLU-Pro","source":{"type":"leaderboard_api","adapter":"openllm"},"what_it_measures_de":"Breites Fachwissen mit zehn Antwortoptionen statt vier und bereinigten Fragen — die Nachfolge fuer das ausgereizte MMLU.","what_it_does_not_measure_de":"Anwendung, Werkzeugnutzung, Codearbeit.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"sample_size":12032,"contamination_risk":"medium","difficulty":"medium","answer_format":"multiple_choice","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":1,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":50,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/mmlu-pro"},{"id":"mmlu","name":"MMLU","domain":"knowledge","subdomain":"broad-academic","url":"https://github.com/hendrycks/test","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"57 Faecher im Multiple-Choice-Format, der lange Zeit meistzitierte Wert.","what_it_does_not_measure_de":"Alles, was ueber Faktenabruf hinausgeht.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"sample_size":14042,"contamination_risk":"high","difficulty":"low","answer_format":"multiple_choice","marketing_weight":"high","notes_de":"Ausgereizt und stark kontaminiert; enthaelt zudem nachweislich fehlerhafte Fragen. Bleibt gelistet, damit aeltere Herstellerangaben einordbar sind.","benchmaxxing_risk":"high","axes":{"recency":0.95,"discrimination":0.56,"headroom":0.41,"contamination":0.15,"benchmaxxing":0.15,"rigor":0.7,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z","top_score":87.8,"top5_spread":6.73,"sample_results":11,"last_result_update":"2026-09-10"},"evidence_score":43,"status":"deprecated","deprecated_by":"mmlu-pro","result_count":11,"url_local":"https://modelradar-one.vercel.app/benchmarks/mmlu"},{"id":"simpleqa","name":"SimpleQA","domain":"knowledge","subdomain":"factuality","url":"https://openai.com/index/introducing-simpleqa/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Kurze Faktenfragen mit eindeutiger Antwort. Misst vor allem, ob ein Modell zugibt, etwas nicht zu wissen, statt zu erfinden.","what_it_does_not_measure_de":"Schliessen, Rechnen, laengere Texte.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":true,"sample_size":4326,"contamination_risk":"medium","difficulty":"high","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.95,"discrimination":1,"headroom":1,"contamination":0.55,"benchmaxxing":0.55,"rigor":0.75,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z","top_score":46.2,"top5_spread":16.1,"sample_results":2,"last_result_update":"2026-09-10"},"evidence_score":81,"status":"active","result_count":2,"url_local":"https://modelradar-one.vercel.app/benchmarks/simpleqa"},{"id":"truthfulqa","name":"TruthfulQA","domain":"safety","subdomain":"factuality","url":"https://github.com/sylinrl/TruthfulQA","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Ob ein Modell verbreiteten Irrtuemern folgt.","what_it_does_not_measure_de":"Fachwissen, aktuelle Fakten.","scale":{"min":0,"max":100,"unit":"% wahrheitsgemaess","higher_is_better":true},"rating":{"test_set_public":true,"sample_size":817,"contamination_risk":"high","difficulty":"low","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":0.55,"rigor":0.7,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":41,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/truthfulqa"},{"id":"ruler","name":"RULER","domain":"long-context","subdomain":"synthetic-retrieval","url":"https://github.com/NVIDIA/RULER","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Bis zu welcher Laenge ein Modell seinen angegebenen Kontext tatsaechlich nutzt. Deckt zuverlaessig auf, wenn ein 1M-Fenster praktisch bei 128K endet.","what_it_does_not_measure_de":"Qualitaet langer Texte, Schliessen ueber Dokumente hinweg.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"llm_judged":false,"contamination_risk":"low","difficulty":"medium","open_harness":true,"scaffold_documented":true,"notes_de":"Aufgaben werden generiert, daher kaum kontaminierbar. Nur mit angegebener Kontextlaenge vergleichbar.","benchmaxxing_risk":"medium","axes":{"recency":0.84,"discrimination":1,"headroom":0.18,"contamination":1,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":1},"computed_at":"2026-09-28T04:00:16.322Z","top_score":94.7,"top5_spread":12.34,"sample_results":2,"last_result_update":"2026-08-01"},"evidence_score":68,"status":"saturated","result_count":2,"url_local":"https://modelradar-one.vercel.app/benchmarks/ruler"},{"id":"longbench-v2","name":"LongBench v2","domain":"long-context","subdomain":"realistic-documents","url":"https://longbench2.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Verstehen und Schliessen ueber echte lange Dokumente statt synthetischer Nadeln.","what_it_does_not_measure_de":"Sehr kurze Aufgaben, Werkzeugnutzung.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"sample_size":503,"contamination_risk":"medium","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.95,"discrimination":1,"headroom":1,"contamination":0.55,"benchmaxxing":1,"rigor":1,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z","top_score":61.9,"top5_spread":21.7,"sample_results":3,"last_result_update":"2026-09-10"},"evidence_score":90,"status":"active","result_count":3,"url_local":"https://modelradar-one.vercel.app/benchmarks/longbench-v2"},{"id":"mrcr","name":"MRCR","domain":"long-context","subdomain":"multi-round-coreference","url":"https://huggingface.co/datasets/openai/mrcr","source":{"type":"manual"},"what_it_measures_de":"Ob ein Modell in langen Dialogen die richtige von mehreren aehnlichen Stellen findet.","what_it_does_not_measure_de":"Faktenwissen.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":53,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/mrcr"},{"id":"babilong","name":"BABILong","domain":"long-context","subdomain":"reasoning-in-haystack","url":"https://github.com/booydar/babilong","source":{"type":"manual"},"what_it_measures_de":"Schliessen ueber verstreute Fakten in sehr langen Texten.","what_it_does_not_measure_de":"Zusammenfassen, Schreiben.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":53,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/babilong"},{"id":"niah","name":"Needle in a Haystack","domain":"long-context","subdomain":"retrieval","url":"https://github.com/gkamradt/LLMTest_NeedleInAHaystack","source":{"type":"manual"},"what_it_measures_de":"Ob ein eingefuegter Satz in einem langen Text wiedergefunden wird.","what_it_does_not_measure_de":"Schliessen, Mehrfachbezuege, Textverstaendnis.","scale":{"min":0,"max":100,"unit":"% gefunden","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"low","difficulty":"low","open_harness":true,"notes_de":"Von aktuellen Modellen praktisch immer geloest. Ein gruenes NIAH-Bild sagt heute fast nichts mehr; RULER oder MRCR sind aussagekraeftiger.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":46,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/niah"},{"id":"mgsm","name":"MGSM","domain":"multilingual","subdomain":"math-transfer","url":"https://huggingface.co/datasets/juletxara/mgsm","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Grundschulmathematik in elf Sprachen — misst Sprachtransfer, nicht Mathematik.","what_it_does_not_measure_de":"Sprachqualitaet, Idiomatik.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"low","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.95,"discrimination":0.26,"headroom":0.52,"contamination":0.15,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z","top_score":84.4,"top5_spread":3.1,"sample_results":2,"last_result_update":"2026-09-10"},"evidence_score":44,"status":"active","result_count":2,"url_local":"https://modelradar-one.vercel.app/benchmarks/mgsm"},{"id":"global-mmlu","name":"Global-MMLU","domain":"multilingual","subdomain":"knowledge","url":"https://huggingface.co/datasets/CohereForAI/Global-MMLU","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"MMLU in 42 Sprachen, kulturell geprueft statt nur maschinell uebersetzt.","what_it_does_not_measure_de":"Generieren in der Zielsprache.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","answer_format":"multiple_choice","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":49,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/global-mmlu"},{"id":"belebele","name":"Belebele","domain":"multilingual","subdomain":"reading-comprehension","url":"https://github.com/facebookresearch/belebele","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Leseverstehen in 122 Sprachvarianten.","what_it_does_not_measure_de":"Generieren, Uebersetzen.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","answer_format":"multiple_choice","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":49,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/belebele"},{"id":"flores-200","name":"FLORES-200","domain":"translation","subdomain":"machine-translation","url":"https://github.com/facebookresearch/flores","source":{"type":"manual"},"what_it_measures_de":"Uebersetzungsqualitaet zwischen 200 Sprachen, gemessen in chrF++/BLEU.","what_it_does_not_measure_de":"Stil, Register, Fachterminologie.","scale":{"min":0,"max":100,"unit":"chrF++","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":55,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/flores-200"},{"id":"include","name":"INCLUDE","domain":"multilingual","subdomain":"regional-knowledge","url":"https://huggingface.co/datasets/CohereForAI/include-base-44","source":{"type":"manual"},"what_it_measures_de":"Regional verankertes Wissen in 44 Sprachen, aus lokalen Pruefungen.","what_it_does_not_measure_de":"Uebersetzung, Sprachqualitaet.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"low","difficulty":"medium","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.322Z"},"evidence_score":57,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/include"},{"id":"mmmu-pro","name":"MMMU-Pro","domain":"vision","subdomain":"expert-multimodal","url":"https://mmmu-benchmark.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Fachaufgaben mit Diagrammen, Tabellen und Skizzen auf Hochschulniveau, mit verschaerftem Aufbau gegen Ratestrategien.","what_it_does_not_measure_de":"Bilderzeugung, Video, reine Texterkennung.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"high","answer_format":"multiple_choice","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.95,"discrimination":1,"headroom":0.87,"contamination":0.55,"benchmaxxing":0.55,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z","top_score":74,"top5_spread":52.2,"sample_results":4,"last_result_update":"2026-09-10"},"evidence_score":80,"status":"active","result_count":4,"url_local":"https://modelradar-one.vercel.app/benchmarks/mmmu-pro"},{"id":"mathvista","name":"MathVista","domain":"vision","subdomain":"visual-math","url":"https://mathvista.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Mathematik, die nur aus dem Bild loesbar ist — Diagramme, Geometrie, Funktionsgraphen.","what_it_does_not_measure_de":"Textmathematik, Beweise.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0,"headroom":1,"contamination":0.33,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.322Z","top_score":43.9,"top5_spread":0,"sample_results":1},"evidence_score":53,"status":"unrated","result_count":1,"url_local":"https://modelradar-one.vercel.app/benchmarks/mathvista"},{"id":"chartqa","name":"ChartQA","domain":"vision","subdomain":"chart-reading","url":"https://github.com/vis-nlp/ChartQA","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Ablesen und Rechnen an Diagrammen.","what_it_does_not_measure_de":"Freie Bildbeschreibung, Video.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"low","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.82,"discrimination":0.18,"headroom":0.53,"contamination":0.15,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.323Z","top_score":83.96,"top5_spread":2.16,"sample_results":2,"last_result_update":"2026-07-25"},"evidence_score":41,"status":"active","result_count":2,"url_local":"https://modelradar-one.vercel.app/benchmarks/chartqa"},{"id":"docvqa","name":"DocVQA","domain":"vision","subdomain":"document-understanding","url":"https://www.docvqa.org/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Fragen zu gescannten Dokumenten, Formularen und Rechnungen.","what_it_does_not_measure_de":"Natuerliche Fotos, Video.","scale":{"min":0,"max":100,"unit":"ANLS","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.95,"discrimination":1,"headroom":0.15,"contamination":0.55,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.323Z","top_score":95.6,"top5_spread":26.3,"sample_results":3,"last_result_update":"2026-09-10"},"evidence_score":71,"status":"saturated","result_count":3,"url_local":"https://modelradar-one.vercel.app/benchmarks/docvqa"},{"id":"blink","name":"BLINK","domain":"vision","subdomain":"visual-perception","url":"https://zeyofu.github.io/blink/","source":{"type":"manual"},"what_it_measures_de":"Wahrnehmungsaufgaben, die Menschen in Sekunden loesen — Tiefe, Spiegelung, Zuordnung. Modelle scheitern hier oft trotz starker Fachwerte.","what_it_does_not_measure_de":"Wissen, Sprache.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.323Z"},"evidence_score":63,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/blink"},{"id":"videomme","name":"Video-MME","domain":"multimodal","subdomain":"video-understanding","url":"https://video-mme.github.io/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Videoverstaendnis ueber kurze bis sehr lange Clips, mit und ohne Untertitel.","what_it_does_not_measure_de":"Standbilder, Audio allein.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.82,"discrimination":0.74,"headroom":1,"contamination":0.55,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.323Z","top_score":59.7,"top5_spread":8.9,"sample_results":2,"last_result_update":"2026-07-25"},"evidence_score":81,"status":"active","result_count":2,"url_local":"https://modelradar-one.vercel.app/benchmarks/videomme"},{"id":"realworldqa","name":"RealWorldQA","domain":"vision","subdomain":"spatial-understanding","url":"https://huggingface.co/datasets/xai-org/RealworldQA","source":{"type":"manual"},"what_it_measures_de":"Raeumliches Verstaendnis in Fotos aus der echten Welt, oft aus Fahrzeugperspektive.","what_it_does_not_measure_de":"Dokumente, Diagramme.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"medium","difficulty":"medium","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.323Z"},"evidence_score":44,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/realworldqa"},{"id":"ai2d","name":"AI2D","domain":"vision","subdomain":"diagram-understanding","url":"https://prior.allenai.org/projects/diagram-understanding","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Verstaendnis von Lehrbuchdiagrammen und deren Beschriftungen.","what_it_does_not_measure_de":"Fotos, Video.","scale":{"min":0,"max":100,"unit":"% korrekt","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"low","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.82,"discrimination":0.29,"headroom":0.62,"contamination":0.15,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.323Z","top_score":81.54,"top5_spread":3.44,"sample_results":2,"last_result_update":"2026-07-25"},"evidence_score":44,"status":"active","result_count":2,"url_local":"https://modelradar-one.vercel.app/benchmarks/ai2d"},{"id":"open-asr-leaderboard","name":"Open ASR Leaderboard","domain":"audio-stt","subdomain":"aggregate-wer","url":"https://huggingface.co/spaces/hf-audio/open_asr_leaderboard","source":{"type":"leaderboard_api","adapter":"openasr"},"what_it_measures_de":"Wortfehlerrate ueber acht englische Datensaetze gemittelt, dazu Geschwindigkeit als Echtzeitfaktor. Die wichtigste vergleichbare Quelle fuer lokale Spracherkennung.","what_it_does_not_measure_de":"Nicht-englische Sprachen, Sprecherzuordnung, Zeichensetzung nach Sinn, Verhalten bei Fachbegriffen und Eigennamen.","scale":{"min":0,"max":100,"unit":"% WER","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"scaffold_documented":true,"notes_de":"Der Mittelwert verdeckt grosse Unterschiede je Datensatz. Fuer eine Kaufentscheidung gehoert der Einzelwert des passenden Datensatzes dazu, nicht nur der Mittelwert.","benchmaxxing_risk":"low","axes":{"recency":1,"discrimination":0.2,"headroom":0.07,"contamination":0.55,"benchmaxxing":1,"rigor":0.8,"reproducibility":1},"computed_at":"2026-09-28T04:00:16.323Z","top_score":7.32,"top5_spread":2.42,"sample_results":21,"last_result_update":"2026-09-28","source_last_update":"2026-09-28"},"evidence_score":58,"status":"active","result_count":21,"url_local":"https://modelradar-one.vercel.app/benchmarks/open-asr-leaderboard"},{"id":"librispeech-test-other","name":"LibriSpeech test-other","domain":"audio-stt","subdomain":"read-speech","url":"https://www.openslr.org/12","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Wortfehlerrate auf vorgelesenen Hoerbuechern, schwierigere Haelfte.","what_it_does_not_measure_de":"Spontansprache, Hintergrundgeraeusche, Akzente ausserhalb der Aufnahme.","scale":{"min":0,"max":100,"unit":"% WER","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"high","difficulty":"low","open_harness":true,"notes_de":"Seit Jahren praktisch ausgereizt und sehr wahrscheinlich in jedem Trainingsdatensatz enthalten. Fuer heutige Entscheidungen kaum noch aussagekraeftig.","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.323Z"},"evidence_score":48,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/librispeech-test-other"},{"id":"common-voice-asr","name":"Common Voice (ASR)","domain":"audio-stt","subdomain":"crowdsourced-accents","url":"https://commonvoice.mozilla.org/","source":{"type":"leaderboard_api","adapter":"openasr"},"what_it_measures_de":"Erkennung ueber viele Sprachen, Akzente und Aufnahmequalitaeten hinweg.","what_it_does_not_measure_de":"Fachvokabular, Mehrsprecher-Situationen.","scale":{"min":0,"max":100,"unit":"% WER","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":1,"discrimination":0.02,"headroom":0.02,"contamination":0.55,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.323Z","top_score":1.8,"top5_spread":0.21,"sample_results":127,"last_result_update":"2026-09-28","source_last_update":"2026-09-28"},"evidence_score":52,"status":"saturated","result_count":127,"url_local":"https://modelradar-one.vercel.app/benchmarks/common-voice-asr"},{"id":"fleurs","name":"FLEURS","domain":"audio-stt","subdomain":"multilingual-asr","url":"https://huggingface.co/datasets/google/fleurs","source":{"type":"leaderboard_api","adapter":"openasr"},"what_it_measures_de":"Spracherkennung in 102 Sprachen, vergleichbar aufgebaut.","what_it_does_not_measure_de":"Spontansprache, Laerm.","scale":{"min":0,"max":100,"unit":"% WER","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":1,"discrimination":0.06,"headroom":0.01,"contamination":0.55,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z","top_score":0.87,"top5_spread":0.69,"sample_results":159,"last_result_update":"2026-09-28","source_last_update":"2026-09-28"},"evidence_score":52,"status":"saturated","result_count":159,"url_local":"https://modelradar-one.vercel.app/benchmarks/fleurs"},{"id":"ami-meeting","name":"AMI Meeting Corpus","domain":"audio-stt","subdomain":"meetings","url":"https://groups.inf.ed.ac.uk/ami/corpus/","source":{"type":"manual"},"what_it_measures_de":"Erkennung in echten Besprechungen mit Ueberlappungen, Nebengeraeuschen und mehreren Sprechern. Der realistischste der gaengigen ASR-Datensaetze.","what_it_does_not_measure_de":"Studioqualitaet, Einzelsprecher.","scale":{"min":0,"max":100,"unit":"% WER","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":63,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/ami-meeting"},{"id":"earnings22","name":"Earnings-22","domain":"audio-stt","subdomain":"domain-specific","url":"https://github.com/revdotcom/speech-datasets","source":{"type":"leaderboard_api","adapter":"openasr"},"what_it_measures_de":"Erkennung in Analystenkonferenzen mit Fachbegriffen, Zahlen und Akzenten.","what_it_does_not_measure_de":"Alltagssprache.","scale":{"min":0,"max":100,"unit":"% WER","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":1,"discrimination":0.26,"headroom":0.1,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z","top_score":9.99,"top5_spread":3.11,"sample_results":21,"last_result_update":"2026-09-28","source_last_update":"2026-09-28"},"evidence_score":68,"status":"active","result_count":21,"url_local":"https://modelradar-one.vercel.app/benchmarks/earnings22"},{"id":"tts-arena","name":"TTS Arena","domain":"audio-tts","subdomain":"human-preference","url":"https://huggingface.co/spaces/TTS-AGI/TTS-Arena","source":{"type":"manual"},"what_it_measures_de":"Menschlicher Blindvergleich zweier Sprachsynthesen, Elo-Wertung.","what_it_does_not_measure_de":"Latenz, Stimmklonen, Sprachabdeckung.","scale":{"min":800,"max":1600,"unit":"Elo","higher_is_better":true},"rating":{"test_set_public":false,"human_verified":true,"llm_judged":false,"contamination_risk":"low","difficulty":"medium","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":63,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/tts-arena"},{"id":"mteb","name":"MTEB","domain":"embedding","subdomain":"aggregate","url":"https://huggingface.co/spaces/mteb/leaderboard","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Einbettungsqualitaet ueber Retrieval, Clustering, Klassifikation und mehr.","what_it_does_not_measure_de":"Generieren, Latenz im Betrieb, Speicherbedarf.","scale":{"min":0,"max":100,"unit":"Ø Score","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"medium","marketing_weight":"high","open_harness":true,"notes_de":"Der Mittelwert ueber sehr verschiedene Teilaufgaben laedt zum Optimieren auf die Rangliste ein. Fuer eine konkrete Anwendung zaehlt die passende Teilaufgabe, nicht der Gesamtwert.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":42,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/mteb"},{"id":"beir","name":"BEIR","domain":"retrieval","subdomain":"zero-shot-retrieval","url":"https://github.com/beir-cellar/beir","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Retrieval ohne aufgabenspezifisches Training ueber 18 Datensaetze.","what_it_does_not_measure_de":"Reranking-Ketten, Latenz.","scale":{"min":0,"max":100,"unit":"nDCG@10","higher_is_better":true},"rating":{"test_set_public":true,"contamination_risk":"high","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":0.55,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":42,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/beir"},{"id":"air-bench","name":"AIR-Bench","domain":"retrieval","subdomain":"fresh-retrieval","url":"https://github.com/AIR-Bench/AIR-Bench","source":{"type":"manual"},"what_it_measures_de":"Retrieval auf laufend erneuerten Korpora, gegen Kontamination gebaut.","what_it_does_not_measure_de":"Klassische statische Vergleiche.","scale":{"min":0,"max":100,"unit":"nDCG@10","higher_is_better":true},"rating":{"test_set_public":false,"private_holdout":true,"contamination_risk":"low","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.5,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":62,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/air-bench"},{"id":"ifeval","name":"IFEval","domain":"instruction-following","subdomain":"verifiable-constraints","url":"https://github.com/google-research/google-research/tree/master/instruction_following_eval","source":{"type":"leaderboard_api","adapter":"openllm"},"what_it_measures_de":"Ob maschinell pruefbare Vorgaben eingehalten werden — Wortzahl, Format, verbotene Woerter. Kein Geschmacksurteil, sondern eine Pruefung.","what_it_does_not_measure_de":"Inhaltliche Qualitaet, Wahrheitsgehalt.","scale":{"min":0,"max":100,"unit":"% eingehalten","higher_is_better":true},"rating":{"test_set_public":true,"llm_judged":false,"sample_size":541,"contamination_risk":"medium","difficulty":"low","open_harness":true,"notes_de":"Objektiv pruefbar, deshalb belastbarer als LLM-bewertete Schreibvergleiche.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.7,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":45,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/ifeval"},{"id":"arena-hard","name":"Arena-Hard","domain":"instruction-following","subdomain":"llm-judged-preference","url":"https://github.com/lmarena/arena-hard-auto","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"500 schwierige Alltagsanfragen, bewertet von einem starken Modell als Richter.","what_it_does_not_measure_de":"Faktentreue, Formattreue, Fachwissen.","scale":{"min":0,"max":100,"unit":"% Siege","higher_is_better":true},"rating":{"test_set_public":true,"llm_judged":true,"sample_size":500,"contamination_risk":"medium","difficulty":"medium","notes_de":"LLM-als-Richter bevorzugt messbar laengere und selbstsicherere Antworten. Nuetzlich als Tendenz, ungeeignet als alleiniges Entscheidungskriterium.","benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.45,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":43,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/arena-hard"},{"id":"alpacaeval-2","name":"AlpacaEval 2.0","domain":"instruction-following","subdomain":"llm-judged-preference","url":"https://tatsu-lab.github.io/alpaca_eval/","source":{"type":"modelcard","adapter":"modelcard"},"what_it_measures_de":"Laengenbereinigte Siegquote gegen ein Referenzmodell, LLM-bewertet.","what_it_does_not_measure_de":"Fachliche Richtigkeit.","scale":{"min":0,"max":100,"unit":"% LC-Siege","higher_is_better":true},"rating":{"test_set_public":true,"llm_judged":true,"sample_size":805,"contamination_risk":"high","difficulty":"low","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.09,"benchmaxxing":0.55,"rigor":0.45,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":39,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/alpacaeval-2"},{"id":"lmarena","name":"LMArena","domain":"instruction-following","subdomain":"human-preference","url":"https://lmarena.ai/","source":{"type":"manual"},"what_it_measures_de":"Blindvergleich durch Menschen ueber sehr viele Anfragen, ausgewertet als Elo. Die groesste Stichprobe echter Nutzerpraeferenz, die es gibt.","what_it_does_not_measure_de":"Richtigkeit, Fachtiefe, Verhalten in Agentenketten. Gemessen wird, was Menschen in einem kurzen Vergleich besser gefaellt.","scale":{"min":800,"max":1600,"unit":"Elo","higher_is_better":true},"rating":{"test_set_public":false,"human_verified":true,"llm_judged":false,"contamination_risk":"low","difficulty":"medium","marketing_weight":"high","notes_de":"Anfaellig fuer Stilpraeferenzen: Formatierung, Laenge und Freundlichkeit wirken stark. Modelle lassen sich gezielt darauf abstimmen, ohne faehiger zu werden.","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":63,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/lmarena"},{"id":"wildbench","name":"WildBench","domain":"instruction-following","subdomain":"real-user-tasks","url":"https://huggingface.co/spaces/allenai/WildBench","source":{"type":"manual"},"what_it_measures_de":"Aufgaben aus echten Nutzerdialogen statt aus einem Aufgabenkatalog.","what_it_does_not_measure_de":"Fachpruefungen, Codeausfuehrung.","scale":{"min":0,"max":100,"unit":"WB-Score","higher_is_better":true},"rating":{"test_set_public":true,"llm_judged":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"medium","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":0.55,"rigor":0.25,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":43,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/wildbench"},{"id":"harmbench","name":"HarmBench","domain":"safety","subdomain":"jailbreak-robustness","url":"https://www.harmbench.org/","source":{"type":"manual"},"what_it_measures_de":"Wie oft standardisierte Angriffe die Schutzmechanismen umgehen.","what_it_does_not_measure_de":"Ueberblockierung harmloser Anfragen.","scale":{"min":0,"max":100,"unit":"% Angriffserfolg","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"medium","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":55,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/harmbench"},{"id":"xstest","name":"XSTest","domain":"safety","subdomain":"over-refusal","url":"https://github.com/paul-rottger/exaggerated-safety","source":{"type":"manual"},"what_it_measures_de":"Wie oft ein Modell harmlose Anfragen faelschlich verweigert. Die Gegenrichtung zu HarmBench, und im Alltag mindestens so wichtig.","what_it_does_not_measure_de":"Echte Schadensrisiken.","scale":{"min":0,"max":100,"unit":"% korrekt beantwortet","higher_is_better":true},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"medium","difficulty":"low","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.33,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":51,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/xstest"},{"id":"jailbreakbench","name":"JailbreakBench","domain":"safety","subdomain":"adaptive-attacks","url":"https://jailbreakbench.github.io/","source":{"type":"manual"},"what_it_measures_de":"Robustheit gegen anpassungsfaehige Angriffe, mit versionierten Artefakten.","what_it_does_not_measure_de":"Alltagsnutzen.","scale":{"min":0,"max":100,"unit":"% Angriffserfolg","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"contamination_risk":"low","difficulty":"high","open_harness":true,"benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.6,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":63,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/jailbreakbench"},{"id":"aa-intelligence-index","name":"Artificial Analysis Intelligence Index","domain":"efficiency","subdomain":"aggregate-index","url":"https://artificialanalysis.ai/","source":{"type":"manual"},"what_it_measures_de":"Zusammengesetzter Index aus mehreren Einzelbenchmarks, dazu unabhaengig gemessene Geschwindigkeit und Preis.","what_it_does_not_measure_de":"Einzelne Faehigkeiten. Ein Index verdeckt genau die Unterschiede, auf die es bei einer konkreten Aufgabe ankommt.","scale":{"min":0,"max":100,"unit":"Indexpunkte","higher_is_better":true},"rating":{"test_set_public":false,"human_verified":false,"contamination_risk":"medium","difficulty":"medium","marketing_weight":"high","notes_de":"Als Ueberblick brauchbar, als Entscheidungsgrundlage fuer eine bestimmte Aufgabe nicht. Die Zusammensetzung aendert sich ueber die Zeit, was Vergleiche zwischen Zeitpunkten erschwert.","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":0.55,"benchmaxxing":1,"rigor":0.5,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":53,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/aa-intelligence-index"},{"id":"throughput-tps","name":"Ausgabegeschwindigkeit (Token/s)","domain":"efficiency","subdomain":"latency","url":"https://artificialanalysis.ai/","source":{"type":"manual"},"what_it_measures_de":"Ausgegebene Token je Sekunde beim jeweiligen Anbieter.","what_it_does_not_measure_de":"Qualitaet. Haengt stark vom Anbieter und der Auslastung ab, nicht nur vom Modell.","scale":{"min":0,"max":1000,"unit":"Token/s","higher_is_better":true},"rating":{"test_set_public":false,"contamination_risk":"low","difficulty":"low","notes_de":"Anbieterabhaengig — ohne Anbieterangabe wertlos.","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.5,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":56,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/throughput-tps"},{"id":"ttft","name":"Zeit bis zum ersten Token","domain":"efficiency","subdomain":"latency","url":"https://artificialanalysis.ai/","source":{"type":"manual"},"what_it_measures_de":"Wartezeit bis zum ersten ausgegebenen Token.","what_it_does_not_measure_de":"Gesamtdauer langer Antworten, Qualitaet.","scale":{"min":0,"max":30,"unit":"Sekunden","higher_is_better":false},"rating":{"test_set_public":false,"contamination_risk":"low","difficulty":"low","benchmaxxing_risk":"low","axes":{"recency":0.5,"discrimination":0.5,"headroom":0.5,"contamination":1,"benchmaxxing":1,"rigor":0.5,"reproducibility":0.2},"computed_at":"2026-09-28T04:00:16.324Z"},"evidence_score":56,"status":"unrated","result_count":0,"url_local":"https://modelradar-one.vercel.app/benchmarks/ttft"},{"id":"tedlium","name":"TED-LIUM 3","domain":"audio-stt","subdomain":"prepared-speech","url":"https://www.openslr.org/51/","source":{"type":"leaderboard_api","adapter":"openasr"},"what_it_measures":"Word error rate on TED talk recordings: prepared, clearly articulated speech.","what_it_measures_de":"Wortfehlerrate auf TED-Vortraegen. Vorbereitete, deutlich gesprochene Rede mit Publikumsgeraeuschen.","what_it_does_not_measure_de":"Spontane Gespraeche, Ueberlappungen, Telefonqualitaet, Fachvokabular.","scale":{"min":0,"max":100,"unit":"% WER","higher_is_better":false},"rating":{"test_set_public":true,"human_verified":true,"llm_judged":false,"contamination_risk":"medium","difficulty":"low","open_harness":true,"notes_de":"Die einfachste der Langform-Aufgaben. Aktuelle Modelle liegen unter 3 % — als Unterscheidungsmerkmal kaum noch tauglich.","top_score":2.12,"top5_spread":0.65,"sample_results":21,"last_result_update":"2026-09-28","benchmaxxing_risk":"low","axes":{"recency":1,"discrimination":0.05,"headroom":0.02,"contamination":0.55,"benchmaxxing":1,"rigor":0.8,"reproducibility":0.6},"computed_at":"2026-09-28T04:00:16.324Z","source_last_update":"2026-09-28"},"evidence_score":49,"status":"saturated","result_count":21,"url_local":"https://modelradar-one.vercel.app/benchmarks/tedlium"}]}