{"sourceCount":139,"providers":[{"slug":"alibaba","label":"Alibaba / Qwen"},{"slug":"anthropic","label":"Anthropic"},{"slug":"arcee","label":"Arcee AI"},{"slug":"bytedance","label":"ByteDance Seed"},{"slug":"cohere","label":"Cohere"},{"slug":"cursor","label":"Cursor"},{"slug":"deepseek","label":"DeepSeek"},{"slug":"google","label":"Google"},{"slug":"inception","label":"Inception Labs"},{"slug":"moonshot","label":"Kimi"},{"slug":"meta","label":"Meta"},{"slug":"minimax","label":"MiniMax"},{"slug":"mistral","label":"Mistral"},{"slug":"nvidia","label":"NVIDIA"},{"slug":"openai","label":"OpenAI"},{"slug":"sakana","label":"Sakana AI"},{"slug":"tencent","label":"Tencent Hunyuan"},{"slug":"thinking-machines","label":"Thinking Machines"},{"slug":"xai","label":"xAI"},{"slug":"z-ai","label":"Z.ai"}],"benchmarks":[{"id":"arc-agi-3","label":"ARC-AGI-3","category":"Reasoning","monitorType":"Verified preview leaderboard","statusNote":"preview evaluation on three unreleased games","sourceUrl":"https://htv3.arc-prize.com/leaderboard","cadenceSeconds":3600},{"id":"cursorbench","label":"CursorBench","category":"Agentic coding","monitorType":"Leaderboard","statusNote":null,"sourceUrl":"https://cursor.com/cursorbench","cadenceSeconds":1800},{"id":"cyberseceval-4","label":"CyberSecEval 4 — Defensive suites","category":"Defensive cybersecurity","monitorType":"Official methodology publication","statusNote":"Publication evidence for AutoPatchBench and CyberSOCEval; no model ranking or offensive-capability recommendation.","sourceUrl":"https://github.com/meta-llama/PurpleLlama/tree/main/CybersecurityBenchmarks","cadenceSeconds":3600},{"id":"deepswe-v1-1","label":"DeepSWE","category":"Agentic coding","monitorType":"Leaderboard","statusNote":null,"sourceUrl":"https://deepswe.datacurve.ai/","cadenceSeconds":1800},{"id":"defenderbench","label":"DefenderBench","category":"Cybersecurity","monitorType":"Official published result table","statusNote":"Mixed offensive, defensive, and cybersecurity-knowledge tasks; rank is derived only from the published overall score.","sourceUrl":"https://github.com/microsoft/DefenderBench#experiment-results","cadenceSeconds":3600},{"id":"design-arena-agentic-web-dev","label":"Design Arena Agentic Web Dev","category":"Design","monitorType":"Leaderboard","statusNote":null,"sourceUrl":"https://designarena.ai/leaderboard","cadenceSeconds":1800},{"id":"frontier-code-1-1-main","label":"FrontierCode","category":"Agentic coding","monitorType":"Leaderboard","statusNote":null,"sourceUrl":"https://cognition.com/frontiercode","cadenceSeconds":1800},{"id":"osworld-verified","label":"OSWorld-Verified","category":"Computer use","monitorType":"Leaderboard","statusNote":null,"sourceUrl":"https://osworld-v1.xlang.ai/","cadenceSeconds":1800},{"id":"swe-bench-multilingual","label":"SWE-bench Multilingual","category":"Coding","monitorType":"Leaderboard","statusNote":null,"sourceUrl":"https://www.swebench.com/","cadenceSeconds":1800},{"id":"swe-bench-verified","label":"SWE-bench Verified","category":"Agentic coding","monitorType":"Leaderboard","statusNote":null,"sourceUrl":"https://www.swebench.com/","cadenceSeconds":1800},{"id":"terminal-bench-3","label":"Terminal-Bench 3","category":"Agentic coding","monitorType":"Official live leaderboard","statusNote":"v3.0 is live with 74 tasks; formerly published as Frontier-Bench","sourceUrl":"https://www.frontierbench.ai/","cadenceSeconds":1800},{"id":"terminal-bench-4","label":"Terminal-Bench 4","category":"Agentic coding","monitorType":"Official live leaderboard","statusNote":"v4.0 is a breaking 66-task cohort with calibrated resources, 19 fixed tasks, and 8 removals","sourceUrl":"https://hub.harborframework.com/datasets/terminal-bench/terminal-bench/latest?tab=leaderboard&leaderboard=4-0-0","cadenceSeconds":14400},{"id":"vocalbench","label":"VocalBench","category":"Vocal conversation","monitorType":"Official GitHub-published leaderboard","statusNote":"Thirteen published semantic, acoustic, chat, latency, and robustness dimensions; per-row architecture and model-weights availability are not published.","sourceUrl":"https://github.com/SJTU-OmniAgent/VocalBench","cadenceSeconds":3600},{"id":"voicebench","label":"VoiceBench","category":"Voice assistants","monitorType":"Official GitHub-published leaderboard","statusNote":"Nine published voice-assistant dimensions; architecture is community-classified on the official page and model weights are marked open or closed.","sourceUrl":"https://matthewcym.github.io/VoiceBench/","cadenceSeconds":3600},{"id":"vulcanbench-v3","label":"VulcanBench Eval Suite 3","category":"Agentic coding","monitorType":"Official machine-readable leaderboard","statusNote":"23 fixed real engineering tasks; each model and reasoning-effort column stays separate","sourceUrl":"https://vulcanbench.com/leaderboard.html","cadenceSeconds":1800}],"targetTypes":{"provider":true,"model_family":false,"benchmark":false}}