{
  "schemaVersion": 1,
  "snapshotDate": "2026-09-15",
  "profileCount": 122,
  "harborCount": 303,
  "huggingFaceCount": 47,
  "recommendationCount": 990,
  "blobfishSuiteCount": 15,
  "provenance": [
    {
      "path": "research/benchmark-company-landscape-2026-09-15/benchmarks.json",
      "sha256": "5aaee4d2e547c29a8a6da82bbb76d2b3b743ec5d0f07b4d230e86fa55cbdfb46"
    },
    {
      "path": "research/benchmark-company-landscape-2026-09-15/harbor-all-303.csv",
      "sha256": "1a3ed7eb72d487636d6e2249d6a694cff701375134a0755fe38562c1ba863985"
    },
    {
      "path": "research/benchmark-company-landscape-2026-09-15/huggingface-official-47.csv",
      "sha256": "faf715a0c9af4ffe87e42d9f946d46e73c22842797dcd228f5e4de322920c20f"
    },
    {
      "path": "research/benchmark-company-landscape-2026-09-15/benchmark-company-matches.csv",
      "sha256": "ec012f346a1cc3fb4a0c9f8b2892990aae1885c89d254c56cfed3546fb14dd1f"
    },
    {
      "path": "research/benchmark-company-landscape-2026-09-15/qualification-additional-matches.csv",
      "sha256": "b5ea3e305dc9e2e208e529c104ae0cc723418f52ebf037540a84d1c9b70d772d"
    },
    {
      "path": "products/website/app/benchmarks/catalog.json",
      "sha256": "c476097114f26cda762077129b64886b6888948eb76a60ff0fd47f6bc9893971"
    }
  ],
  "benchmarks": [
    {
      "id": "webvoyager",
      "profileId": "B001",
      "name": "WebVoyager",
      "category": "Browser automation",
      "description": "End-to-end navigation and task completion on live websites; original release has 643 tasks across 15 sites.",
      "metrics": "Task success; supplemented with cost per success and elapsed time.",
      "execution": "Browser agent with live-web access and recorded trajectories.",
      "limitations": "Website drift, blocked pages and judge configuration affect comparability; distinguish the dataset from its reference agent.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/MinorJerry/WebVoyager",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Browser Use",
          "url": "https://browser-use.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Browserbase / Stagehand",
          "url": "https://www.browserbase.com/blog/introducing-the-stagehand-api",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Firecrawl",
          "url": "https://www.firecrawl.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Microsoft Playwright",
          "url": "https://playwright.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Skyvern",
          "url": "https://www.skyvern.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Steel",
          "url": "https://steel.dev/",
          "fit": "Composite system"
        },
        {
          "name": "TinyFish",
          "url": "https://docs.tinyfish.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "online-mind2web",
      "profileId": "B002",
      "name": "Online-Mind2Web",
      "category": "Browser automation",
      "description": "300 real-world tasks spanning 136 websites; maintainers update invalid tasks.",
      "metrics": "Task success under the pinned evaluation protocol; human review or specified WebJudge.",
      "execution": "Live browser, action/screenshot trace and dated task snapshot.",
      "limitations": "Pin updated task IDs and evaluator. It is different from offline Mind2Web and Mind2Web 2.",
      "priority": "P1",
      "maturity": "Established and maintained",
      "sourceUrl": "https://github.com/OSU-NLP-Group/Online-Mind2Web",
      "catalogUrl": "https://huggingface.co/datasets/osunlp/Online-Mind2Web",
      "companies": [
        {
          "name": "Browser Use",
          "url": "https://browser-use.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Browserbase / Stagehand",
          "url": "https://www.browserbase.com/blog/introducing-the-stagehand-api",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Firecrawl",
          "url": "https://www.firecrawl.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Microsoft Playwright",
          "url": "https://playwright.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Skyvern",
          "url": "https://www.skyvern.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Steel",
          "url": "https://steel.dev/",
          "fit": "Composite system"
        },
        {
          "name": "TinyFish",
          "url": "https://docs.tinyfish.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "webarena",
      "profileId": "B003",
      "name": "WebArena",
      "category": "Browser automation",
      "description": "Multi-step commerce, forum, CMS and developer-site workflows in resettable websites.",
      "metrics": "Execution-based task success.",
      "execution": "Self-hosted sites plus BrowserGym or original harness.",
      "limitations": "Preserve original site data and resets; a live-web implementation is a different benchmark.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/web-arena-x/webarena",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Automation Anywhere",
          "url": "https://www.automationanywhere.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Browser Use",
          "url": "https://browser-use.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Browserbase / Stagehand",
          "url": "https://www.browserbase.com/blog/introducing-the-stagehand-api",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Firecrawl",
          "url": "https://www.firecrawl.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Microsoft Playwright",
          "url": "https://playwright.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Skyvern",
          "url": "https://www.skyvern.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Steel",
          "url": "https://steel.dev/",
          "fit": "Composite system"
        },
        {
          "name": "TinyFish",
          "url": "https://docs.tinyfish.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "visualwebarena",
      "profileId": "B004",
      "name": "VisualWebArena",
      "category": "Browser automation",
      "description": "Browser tasks requiring visual understanding, including images and visually grounded shopping.",
      "metrics": "Execution-based task success.",
      "execution": "Screenshot-capable agent plus the benchmark websites.",
      "limitations": "DOM-only agents do not exercise the same visual task setting; image access must be declared.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/web-arena-x/visualwebarena",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Browser Use",
          "url": "https://browser-use.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Browserbase / Stagehand",
          "url": "https://www.browserbase.com/blog/introducing-the-stagehand-api",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Skyvern",
          "url": "https://www.skyvern.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "TinyFish",
          "url": "https://docs.tinyfish.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "workarena-workarena",
      "profileId": "B005",
      "name": "WorkArena / WorkArena++",
      "category": "Browser automation",
      "description": "ServiceNow knowledge-work tasks and compositional enterprise workflows.",
      "metrics": "Task success, reported separately by task level.",
      "execution": "ServiceNow instance and BrowserGym integration.",
      "limitations": "An instance and supported configuration are needed; browser workflow performance does not measure every enterprise integration.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/ServiceNow/WorkArena",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Automation Anywhere",
          "url": "https://www.automationanywhere.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Browser Use",
          "url": "https://browser-use.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Browserbase / Stagehand",
          "url": "https://www.browserbase.com/blog/introducing-the-stagehand-api",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ServiceNow",
          "url": "https://www.servicenow.com/products/ai-agents.html",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Skyvern",
          "url": "https://www.skyvern.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "osworld-verified",
      "profileId": "B006",
      "name": "OSWorld-Verified",
      "category": "Desktop and mobile agents",
      "description": "Desktop tasks involving applications, files and cross-application workflows.",
      "metrics": "Final environment-state task success.",
      "execution": "Desktop VM, screenshot/input adapter and pinned OS image.",
      "limitations": "Harbor has 361 tasks and excludes eight Google Drive/login tasks; disclose this versus the original 369-task suite.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/xlang-ai/OSWorld",
      "catalogUrl": "https://hub.harborframework.com/datasets/xlang-ai/osworld-verified",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Automation Anywhere",
          "url": "https://www.automationanywhere.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "xlang-ai/osworld-verified",
          "url": "https://hub.harborframework.com/datasets/xlang-ai/osworld-verified",
          "tasks": 361,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "osworld-2-0",
      "profileId": "B007",
      "name": "OSWorld 2.0",
      "category": "Desktop and mobile agents",
      "description": "Long-horizon computer-use tasks with coordinated code, assets and mocked websites.",
      "metrics": "Task completion under the version-specific evaluator.",
      "execution": "Desktop runtime and release-matched gated assets.",
      "limitations": "Pin code, task files, assets and website release together; OSWorld 1.x scores are not directly comparable.",
      "priority": "P2",
      "maturity": "Newer extension",
      "sourceUrl": "https://github.com/xlang-ai/OSWorld-V2",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Automation Anywhere",
          "url": "https://www.automationanywhere.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "screenspot-pro",
      "profileId": "B008",
      "name": "ScreenSpot-Pro",
      "category": "Desktop and mobile agents",
      "description": "Locate the correct UI target in professional, high-resolution screenshots.",
      "metrics": "Grounding / click-point accuracy.",
      "execution": "Model returning screen coordinates for image and instruction.",
      "limitations": "A component test: a correct click does not establish multi-step task success.",
      "priority": "P2",
      "maturity": "Established specialist",
      "sourceUrl": "https://huggingface.co/datasets/likaixin/ScreenSpot-Pro",
      "catalogUrl": "https://huggingface.co/datasets/likaixin/ScreenSpot-Pro",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Automation Anywhere",
          "url": "https://www.automationanywhere.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "likaixin/ScreenSpot-Pro",
          "url": "https://huggingface.co/datasets/likaixin/ScreenSpot-Pro",
          "tasks": null,
          "access": "Public",
          "review": "Selected GUI-grounding component test."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "androidworld",
      "profileId": "B009",
      "name": "AndroidWorld",
      "category": "Desktop and mobile agents",
      "description": "116 parameterized task templates across 20 Android apps.",
      "metrics": "Environment-state task success.",
      "execution": "Android emulator, supported apps and action adapter.",
      "limitations": "Only applicable to products able to control Android; emulator tasks do not establish iOS compatibility.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/google-research/android_world",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Samsung",
          "url": "https://www.samsung.com/ae/news/local/galaxy-ai-expands-multi-agent-ecosystem-to-give-users-more-choice-and-flexibility/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "swe-bench-verified",
      "profileId": "B010",
      "name": "SWE-bench Verified",
      "category": "Coding and terminal agents",
      "description": "Resolve 500 human-validated GitHub issues.",
      "metrics": "Percentage resolved by the official test harness.",
      "execution": "Repository-editing agent and isolated executable tests.",
      "limitations": "Hold repository commit, dependency images, attempts and agent scaffold fixed; historical public tasks can be contaminated.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Verified",
      "catalogUrl": "https://hub.harborframework.com/datasets/swe-bench/swe-bench-verified",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "swe-bench/swe-bench-verified",
          "url": "https://hub.harborframework.com/datasets/swe-bench/swe-bench-verified",
          "tasks": 500,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "swe-bench-pro",
      "profileId": "B011",
      "name": "SWE-bench Pro",
      "category": "Coding and terminal agents",
      "description": "Long-horizon repository-level software tasks; public Harbor snapshot has 731 tasks.",
      "metrics": "Resolved-task rate.",
      "execution": "Repository agent and Pro-specific environments/tests.",
      "limitations": "Use the public split identity; do not treat Pro, Verified and private commercial evaluations as interchangeable.",
      "priority": "P1",
      "maturity": "Established newer suite",
      "sourceUrl": "https://github.com/scaleapi/SWE-bench_Pro-os",
      "catalogUrl": "https://huggingface.co/datasets/ScaleAI/SWE-bench_Pro",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sourcegraph",
          "url": "https://sourcegraph.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "ScaleAI/SWE-bench_Pro",
          "url": "https://huggingface.co/datasets/ScaleAI/SWE-bench_Pro",
          "tasks": null,
          "access": "Public",
          "review": "Selected repository-level coding suite."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "swe-bench-multilingual",
      "profileId": "B012",
      "name": "SWE-bench Multilingual",
      "category": "Coding and terminal agents",
      "description": "Repository issue resolution across nine programming languages, 300 test instances.",
      "metrics": "Percentage resolved, with language breakdowns.",
      "execution": "Language-specific build environments and agent adapter.",
      "limitations": "Multilingual means programming languages here, not human-language support.",
      "priority": "P1",
      "maturity": "Established extension",
      "sourceUrl": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Multilingual",
      "catalogUrl": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Multilingual",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tabnine",
          "url": "https://www.tabnine.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "SWE-bench/SWE-bench_Multilingual",
          "url": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Multilingual",
          "tasks": null,
          "access": "Public",
          "review": "Selected programming-language coverage extension."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "terminal-bench-2-1",
      "profileId": "B013",
      "name": "Terminal-Bench 2.1",
      "category": "Coding and terminal agents",
      "description": "89 difficult terminal tasks across engineering, systems and scientific work.",
      "metrics": "Verifier-defined task success.",
      "execution": "Containerized command-line agent, Harbor and fixed resource budgets.",
      "limitations": "Keep 2.1 as a historical comparison track; it is not the latest release. Do not mix its results with 2.0.",
      "priority": "P1",
      "maturity": "Established comparison baseline",
      "sourceUrl": "https://www.tbench.ai/news/terminal-bench-2-1",
      "catalogUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2-1",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "terminal-bench/terminal-bench-2-1",
          "url": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2-1",
          "tasks": 89,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "terminal-bench-4-0",
      "profileId": "B014",
      "name": "Terminal-Bench 4.0",
      "category": "Coding and terminal agents",
      "description": "Current continuous Terminal-Bench release with task and resource revisions; catalog snapshot lists 66 tasks.",
      "metrics": "Verifier-defined success under the versioned runtime budget.",
      "execution": "Harbor terminal agent; some tasks need GPUs or multiple containers.",
      "limitations": "The release sets an eight-hour agent timeout. Use the exact v4.0.0 task manifest; a shorter-budget run needs its own label. 3.0 and 2.1 remain separate.",
      "priority": "P1",
      "maturity": "Current release; higher run cost",
      "sourceUrl": "https://www.tbench.ai/news/terminal-bench-4-0",
      "catalogUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "terminal-bench/terminal-bench",
          "url": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench",
          "tasks": 66,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "livecodebench",
      "profileId": "B015",
      "name": "LiveCodeBench",
      "category": "Coding and terminal agents",
      "description": "Competition-style coding, with date-bounded problem releases.",
      "metrics": "Pass@1 / pass@k on the declared task window.",
      "execution": "Code generation and isolated test execution.",
      "limitations": "Select problems newer than the evaluated model training cutoff where possible. Harbor has a 100-task port, not the entire evolving dataset.",
      "priority": "P1",
      "maturity": "Established and refreshed",
      "sourceUrl": "https://github.com/LiveCodeBench/LiveCodeBench",
      "catalogUrl": "https://huggingface.co/datasets/livecodebench/code_generation_lite",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "aider-polyglot",
      "profileId": "B016",
      "name": "Aider Polyglot",
      "category": "Coding and terminal agents",
      "description": "225 programming exercises across six languages.",
      "metrics": "Exercise pass rate and editing-format success.",
      "execution": "Aider-style edit/apply/test interaction.",
      "limitations": "Useful for editing and language coverage; does not replace repository-scale SWE evaluation. Keep retry policy fixed.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://aider.chat/docs/leaderboards/",
      "catalogUrl": "https://hub.harborframework.com/datasets/aider/aider-polyglot",
      "companies": [
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tabnine",
          "url": "https://www.tabnine.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "aider/aider-polyglot",
          "url": "https://hub.harborframework.com/datasets/aider/aider-polyglot",
          "tasks": 225,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "bigcodebench-bigcodebench-hard",
      "profileId": "B017",
      "name": "BigCodeBench / BigCodeBench-Hard",
      "category": "Coding and terminal agents",
      "description": "Python programming tasks using real libraries and APIs.",
      "metrics": "Pass@1 using executable tests.",
      "execution": "Pinned Python packages and code execution.",
      "limitations": "Full, Hard, Complete and Instruct tracks differ; Harbor Hard-Complete lists 145 tasks, so inspect membership before comparison.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/bigcode-project/bigcodebench",
      "catalogUrl": "https://huggingface.co/datasets/bigcode/bigcodebench-hard",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "humaneval-mbpp-evalplus",
      "profileId": "B018",
      "name": "HumanEval+ / MBPP+ (EvalPlus)",
      "category": "Coding and terminal agents",
      "description": "Small Python function synthesis with expanded tests.",
      "metrics": "Pass@1 / pass@k.",
      "execution": "Lightweight isolated Python evaluation.",
      "limitations": "Good regression checks, weak standalone evidence for modern autonomous coding; public-data exposure and saturation limit differentiation.",
      "priority": "P3",
      "maturity": "Historical baseline",
      "sourceUrl": "https://github.com/evalplus/evalplus",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "deepswe-1-1",
      "profileId": "B019",
      "name": "DeepSWE 1.1",
      "category": "Coding and terminal agents",
      "description": "113 original long-horizon repository tasks across five programming languages.",
      "metrics": "Behavioral verifier success.",
      "execution": "Harbor with a separate pristine verifier environment.",
      "limitations": "Pin 1.1 and verifier runtime; the reference patch is not the grading criterion. Do not confuse this benchmark with similarly named models.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://hub.harborframework.com/datasets/datacurve/deep-swe-1-1",
      "catalogUrl": "https://huggingface.co/datasets/datacurve/deep-swe",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sourcegraph",
          "url": "https://sourcegraph.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "datacurve/deep-swe",
          "url": "https://huggingface.co/datasets/datacurve/deep-swe",
          "tasks": null,
          "access": "Gated",
          "review": "Selected newer long-horizon coding family; pin 1.1 versus prior release."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "swe-atlas-qna-test-writing-refactoring",
      "profileId": "B020",
      "name": "SWE-Atlas: QnA / Test Writing / Refactoring",
      "category": "Coding and terminal agents",
      "description": "Codebase understanding, test authoring and refactoring as separate tracks.",
      "metrics": "Track-specific correctness and test-quality measures.",
      "execution": "Repository agent and separate Atlas harness tracks.",
      "limitations": "Report each track separately. Harbor lists 124 QnA, 90 test-writing and 70 refactoring tasks.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://github.com/scaleapi/SWE-Atlas",
      "catalogUrl": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-qna",
      "companies": [
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sourcegraph",
          "url": "https://sourcegraph.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "scale-ai/swe-atlas-qna",
          "url": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-qna",
          "tasks": 124,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "android-bench",
      "profileId": "B021",
      "name": "Android Bench",
      "category": "Coding and terminal agents",
      "description": "100 software engineering tasks on Android repositories.",
      "metrics": "Build/test-based task correctness.",
      "execution": "KVM-capable Linux host, Android tooling and emulators.",
      "limitations": "This tests building/fixing Android software, not using phone apps. The documented local minimum is 32 GB RAM and 120 GB disk.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://hub.harborframework.com/datasets/android-bench/android-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/android-bench/android-bench",
      "companies": [
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "android-bench/android-bench",
          "url": "https://hub.harborframework.com/datasets/android-bench/android-bench",
          "tasks": 100,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "long-horizon-terminal-bench",
      "profileId": "B022",
      "name": "Long-Horizon Terminal-Bench",
      "category": "Coding and terminal agents",
      "description": "46 tasks testing sustained useful work over hundreds of steps.",
      "metrics": "Hidden verifier success after rebuilding from the output artifact.",
      "execution": "Long-running terminal agent with stateful workspaces.",
      "limitations": "Long-running persistence test; do not confuse the LHTB publication with a numbered Terminal-Bench release.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/IntelligenceLab/Long-Horizon-Terminal-Bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/long-horizon-terminal-bench/lhtb",
      "companies": [
        {
          "name": "All Hands / OpenHands",
          "url": "https://www.all-hands.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "JetBrains",
          "url": "https://www.jetbrains.com/junie/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Replit",
          "url": "https://replit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "long-horizon-terminal-bench/lhtb",
          "url": "https://hub.harborframework.com/datasets/long-horizon-terminal-bench/lhtb",
          "tasks": 46,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "bench-bench",
      "profileId": "B023",
      "name": "τ-bench / τ²-bench",
      "category": "Customer service and tool use",
      "description": "Policy-following conversations, tool actions and cooperative customer-service tasks.",
      "metrics": "Task success and pass^k consistency across repeated runs.",
      "execution": "Agent linked to benchmark tools and a fixed user simulator.",
      "limitations": "pass^k measures consistent success and is not pass@k. Pin user simulator, policy version and turn limits.",
      "priority": "P1",
      "maturity": "Established comparison family",
      "sourceUrl": "https://arxiv.org/abs/2506.07982",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Ada",
          "url": "https://docs.ada.cx/docs/welcome",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Decagon",
          "url": "https://decagon.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Forethought",
          "url": "https://forethought.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Intercom",
          "url": "https://www.intercom.com/fin",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "ServiceNow",
          "url": "https://www.servicenow.com/products/ai-agents.html",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Sierra",
          "url": "https://sierra.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zendesk",
          "url": "https://developer.zendesk.com/documentation/ai-agents/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "bench-text-and-knowledge",
      "profileId": "B024",
      "name": "τ³-bench: text and knowledge",
      "category": "Customer service and tool use",
      "description": "Customer service in airline, retail, telecom and banking-knowledge domains; Harbor lists 375 tasks.",
      "metrics": "Domain-specific reward and repeated-run reliability.",
      "execution": "Tool adapter plus user simulation; retrieval setup for knowledge tasks.",
      "limitations": "Banking grading changed in v1.0.1; pin task/evaluator release. The Harbor listing alone does not establish native audio support.",
      "priority": "P1",
      "maturity": "Current extension",
      "sourceUrl": "https://github.com/sierra-research/tau2-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/sierra-research/tau3-bench",
      "companies": [
        {
          "name": "Ada",
          "url": "https://docs.ada.cx/docs/welcome",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Decagon",
          "url": "https://decagon.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Forethought",
          "url": "https://forethought.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "HubSpot",
          "url": "https://knowledge.hubspot.com/ai/understand-breeze",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Intercom",
          "url": "https://www.intercom.com/fin",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "ServiceNow",
          "url": "https://www.servicenow.com/products/ai-agents.html",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Sierra",
          "url": "https://sierra.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zendesk",
          "url": "https://developer.zendesk.com/documentation/ai-agents/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "sierra-research/tau3-bench",
          "url": "https://hub.harborframework.com/datasets/sierra-research/tau3-bench",
          "tasks": 375,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "berkeley-function-calling-leaderboard-bfcl",
      "profileId": "B025",
      "name": "Berkeley Function Calling Leaderboard (BFCL)",
      "category": "Customer service and tool use",
      "description": "Function selection, argument generation, multi-turn tool use and current agentic tracks.",
      "metrics": "Category-specific function-call correctness.",
      "execution": "Tool-call model adapter or instrumented agent interface.",
      "limitations": "Separate model-only calling quality from a complete product agent. Report BFCL version and categories, not an unspecified aggregate.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://gorilla.cs.berkeley.edu/leaderboard",
      "catalogUrl": "https://hub.harborframework.com/datasets/gorilla/bfcl",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Composio",
          "url": "https://composio.dev/",
          "fit": "Composite system"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zapier",
          "url": "https://zapier.com/agents",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "n8n",
          "url": "https://n8n.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "gorilla/bfcl",
          "url": "https://hub.harborframework.com/datasets/gorilla/bfcl",
          "tasks": 3641,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "toolsandbox",
      "profileId": "B026",
      "name": "ToolSandbox",
      "category": "Customer service and tool use",
      "description": "Stateful conversations with dependent tools and implicit contextual requirements.",
      "metrics": "Milestone / state-based task success.",
      "execution": "Adapter to benchmark tools and world state.",
      "limitations": "Requires preservation of state and tool semantics; generic chat-only evaluation is insufficient.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/apple-aiml-research/ToolSandbox",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Apple",
          "url": "https://developer.apple.com/apple-intelligence/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Composio",
          "url": "https://composio.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lindy",
          "url": "https://www.lindy.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zapier",
          "url": "https://zapier.com/agents",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "appworld",
      "profileId": "B027",
      "name": "AppWorld",
      "category": "Customer service and tool use",
      "description": "Personal-assistant workflows spanning simulated applications and programmatic APIs.",
      "metrics": "Task goal completion and unintended-change checks.",
      "execution": "AppWorld runtime plus interactive code/tool agent.",
      "limitations": "Map tools into the simulated apps; access to a real SaaS connector does not make the product directly runnable.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/StonyBrookNLP/appworld",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Composio",
          "url": "https://composio.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Gumloop",
          "url": "https://www.gumloop.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lindy",
          "url": "https://www.lindy.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Zapier",
          "url": "https://zapier.com/agents",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "n8n",
          "url": "https://n8n.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "toolathlon-verified",
      "profileId": "B028",
      "name": "Toolathlon-Verified",
      "category": "Customer service and tool use",
      "description": "108 verified long-horizon tasks across real software tool environments.",
      "metrics": "Task-specific outcome verification.",
      "execution": "Release-specific tools, apps and agent adapter.",
      "limitations": "Gated dataset and environment dependencies require setup; evaluate the verified task release, not an arbitrary older copy.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/hkust-nlp/Toolathlon",
      "catalogUrl": "https://huggingface.co/datasets/hkust-nlp/Toolathlon",
      "companies": [
        {
          "name": "Composio",
          "url": "https://composio.dev/",
          "fit": "Composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Gumloop",
          "url": "https://www.gumloop.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lindy",
          "url": "https://www.lindy.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Zapier",
          "url": "https://zapier.com/agents",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "n8n",
          "url": "https://n8n.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "hkust-nlp/Toolathlon",
          "url": "https://huggingface.co/datasets/hkust-nlp/Toolathlon",
          "tasks": null,
          "access": "Gated",
          "review": "Selected verified long-horizon tool-use suite; gated."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "crmarena-crmarena-pro",
      "profileId": "B029",
      "name": "CRMArena / CRMArena-Pro",
      "category": "Enterprise work and professional deliverables",
      "description": "CRM analysis and actions in realistic sales, service and CPQ scenarios.",
      "metrics": "Task correctness and conversation/workflow measures.",
      "execution": "Salesforce sandbox/API adapter.",
      "limitations": "Salesforce schemas need deliberate mapping for other CRMs. GUI access now requires a request; do not assume public GUI credentials work.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/SalesforceAIResearch/CRMArena",
      "catalogUrl": "https://huggingface.co/datasets/Salesforce/CRMArenaPro",
      "companies": [
        {
          "name": "Decagon",
          "url": "https://decagon.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Gumloop",
          "url": "https://www.gumloop.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "HubSpot",
          "url": "https://knowledge.hubspot.com/ai/understand-breeze",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Lindy",
          "url": "https://www.lindy.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Sierra",
          "url": "https://sierra.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "theagentcompany",
      "profileId": "B030",
      "name": "TheAgentCompany",
      "category": "Enterprise work and professional deliverables",
      "description": "174 simulated company tasks across files, chat, project management and development tools.",
      "metrics": "Task success / benchmark-defined partial credit.",
      "execution": "Multi-service company environment and agent tools.",
      "limitations": "Higher hosting/setup effort; report final work artifacts and application state, not just plausible chat replies.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/TheAgentCompany/TheAgentCompany",
      "catalogUrl": "https://hub.harborframework.com/datasets/theagentcompany/theagentcompany",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Automation Anywhere",
          "url": "https://www.automationanywhere.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Gumloop",
          "url": "https://www.gumloop.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lindy",
          "url": "https://www.lindy.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "ServiceNow",
          "url": "https://www.servicenow.com/products/ai-agents.html",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zapier",
          "url": "https://zapier.com/agents",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "n8n",
          "url": "https://n8n.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "theagentcompany/theagentcompany",
          "url": "https://hub.harborframework.com/datasets/theagentcompany/theagentcompany",
          "tasks": 174,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "apex-agents-1-1",
      "profileId": "B031",
      "name": "APEX-Agents 1.1",
      "category": "Enterprise work and professional deliverables",
      "description": "240 tasks in investment banking, consulting and law in the Harbor 1.1 package.",
      "metrics": "Binary rubric criteria and all-criteria task success.",
      "execution": "File and application tools plus the published rubric evaluator.",
      "limitations": "The HF card currently describes 480 tasks; Harbor 1.1 has 240. Pin version/split. Intended for evaluation only; no training reuse.",
      "priority": "P2",
      "maturity": "Newer professional suite",
      "sourceUrl": "https://hub.harborframework.com/datasets/mercor/apex-agents-1-1",
      "catalogUrl": "https://huggingface.co/datasets/mercor/apex-agents",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Microsoft Copilot",
          "url": "https://www.microsoft.com/en-us/microsoft-365-copilot",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "mercor/apex-agents",
          "url": "https://huggingface.co/datasets/mercor/apex-agents",
          "tasks": null,
          "access": "Gated",
          "review": "Selected professional-agent family; HF 480-task description differs from Harbor 1.1 240-task snapshot."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "skillsbench-v1-1",
      "profileId": "B032",
      "name": "SkillsBench v1.1",
      "category": "Enterprise work and professional deliverables",
      "description": "87-task snapshot for testing the contribution of agent skills.",
      "metrics": "Task reward; compare with and without the supplied skills.",
      "execution": "Identical agent/model/runtime in paired skill conditions.",
      "limitations": "A skills-effect experiment requires a paired baseline; an absolute score alone does not prove that skills helped.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/benchflow/skillsbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/benchflow/skillsbench",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cline",
          "url": "https://cline.bot/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Gumloop",
          "url": "https://www.gumloop.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lindy",
          "url": "https://www.lindy.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "benchflow/skillsbench",
          "url": "https://hub.harborframework.com/datasets/benchflow/skillsbench",
          "tasks": 87,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-index-1-0",
      "profileId": "B033",
      "name": "Harbor Index 1.0",
      "category": "Enterprise work and professional deliverables",
      "description": "82 tasks curated from a larger candidate pool for broad agentic evaluation.",
      "metrics": "Versioned aggregate reward and per-capability breakdown.",
      "execution": "Harbor-supported agent.",
      "limitations": "Useful cross-capability screen; inspect overlap before combining with constituent benchmarks. Not a replacement for vertical depth.",
      "priority": "P2",
      "maturity": "Newer broad diagnostic",
      "sourceUrl": "https://hub.harborframework.com/datasets/harbor-index/harbor-index-1.0",
      "catalogUrl": "https://hub.harborframework.com/datasets/harbor-index/harbor-index-1.0",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cognition / Devin",
          "url": "https://cognition.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cursor",
          "url": "https://cursor.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "harbor-index/harbor-index-1.0",
          "url": "https://hub.harborframework.com/datasets/harbor-index/harbor-index-1.0",
          "tasks": 82,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "gaia",
      "profileId": "B034",
      "name": "GAIA",
      "category": "Search and deep research",
      "description": "Multi-step questions requiring tools, browsing, files and sometimes multimodal inputs.",
      "metrics": "Final-answer accuracy by difficulty level.",
      "execution": "General agent with the required tool and media access.",
      "limitations": "Gated access; the Harbor package has 165 tasks, a public validation subset rather than the entire hidden evaluation.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/gaia-benchmark/GAIA",
      "catalogUrl": "https://hub.harborframework.com/datasets/gaia/gaia",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Exa",
          "url": "https://exa.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Parallel",
          "url": "https://parallel.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Perplexity",
          "url": "https://www.perplexity.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tavily",
          "url": "https://www.tavily.com/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "You.com",
          "url": "https://you.com/",
          "fit": "Research endpoint or composite system"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "gaia/gaia",
          "url": "https://hub.harborframework.com/datasets/gaia/gaia",
          "tasks": 165,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "browsecomp",
      "profileId": "B035",
      "name": "BrowseComp",
      "category": "Search and deep research",
      "description": "1,266 hard-to-find web information questions.",
      "metrics": "Answer correctness under the original evaluator.",
      "execution": "Search/browsing agent with fixed budgets.",
      "limitations": "Measures persistent fact-finding; does not establish report quality, citation quality or transaction execution. Search APIs need a fixed agent wrapper.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://openai.com/index/browsecomp/",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Exa",
          "url": "https://exa.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Parallel",
          "url": "https://parallel.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Perplexity",
          "url": "https://www.perplexity.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tavily",
          "url": "https://www.tavily.com/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "You.com",
          "url": "https://you.com/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Firecrawl",
          "url": "https://www.firecrawl.dev/",
          "fit": "Conditional on qualifying the research-preview Agent as an answer-producing system; raw scraping alone does not qualify."
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "deepresearchbench",
      "profileId": "B036",
      "name": "DeepResearchBench",
      "category": "Search and deep research",
      "description": "Long-form research report generation on 100 research tasks.",
      "metrics": "RACE report-quality and FACT citation/factuality evaluations.",
      "execution": "Research system returning reports and citations; fixed judge pipeline.",
      "limitations": "The official evaluator changed in 2026. Pin judge and reference reports; results under different judges are not interchangeable.",
      "priority": "P1",
      "maturity": "Established newer suite",
      "sourceUrl": "https://github.com/Ayanami0730/deep_research_bench",
      "catalogUrl": "https://huggingface.co/datasets/muset-ai/DeepResearch-Bench-Dataset",
      "companies": [
        {
          "name": "AlphaSense",
          "url": "https://www.alpha-sense.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Consensus",
          "url": "https://consensus.app/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Elicit",
          "url": "https://elicit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Exa",
          "url": "https://exa.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Parallel",
          "url": "https://parallel.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Perplexity",
          "url": "https://www.perplexity.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tavily",
          "url": "https://www.tavily.com/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "You.com",
          "url": "https://you.com/",
          "fit": "Research endpoint or composite system"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mind2web-2",
      "profileId": "B037",
      "name": "Mind2Web 2",
      "category": "Search and deep research",
      "description": "Long-horizon web research and information synthesis.",
      "metrics": "Agent-as-a-Judge task evaluation.",
      "execution": "Research agent and the benchmark judge workflow.",
      "limitations": "Different from Online-Mind2Web navigation. Report judge model and source/date conditions.",
      "priority": "P2",
      "maturity": "Established newer suite",
      "sourceUrl": "https://github.com/OSU-NLP-Group/Mind2Web-2",
      "catalogUrl": "https://huggingface.co/datasets/osunlp/Mind2Web-2",
      "companies": [
        {
          "name": "Consensus",
          "url": "https://consensus.app/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Elicit",
          "url": "https://elicit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Exa",
          "url": "https://exa.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Parallel",
          "url": "https://parallel.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Perplexity",
          "url": "https://www.perplexity.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tavily",
          "url": "https://www.tavily.com/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "You.com",
          "url": "https://you.com/",
          "fit": "Research endpoint or composite system"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "frames",
      "profileId": "B038",
      "name": "FRAMES",
      "category": "Search and deep research",
      "description": "Multi-hop factual retrieval and reasoning across source documents.",
      "metrics": "Answer accuracy, with reasoning-type slices.",
      "execution": "RAG/research pipeline with fixed retrieval conditions.",
      "limitations": "A useful diagnostic for synthesis, not proof of broad open-web coverage or private enterprise connector quality.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/google/frames-benchmark",
      "catalogUrl": "https://huggingface.co/datasets/google/frames-benchmark",
      "companies": [
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Exa",
          "url": "https://exa.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Parallel",
          "url": "https://parallel.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Perplexity",
          "url": "https://www.perplexity.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tavily",
          "url": "https://www.tavily.com/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Vectara",
          "url": "https://www.vectara.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "You.com",
          "url": "https://you.com/",
          "fit": "Research endpoint or composite system"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "simpleqa",
      "profileId": "B039",
      "name": "SimpleQA",
      "category": "Search and deep research",
      "description": "Short fact-seeking questions with reference answers.",
      "metrics": "Correctness, abstention and attempted-answer accuracy.",
      "execution": "Question-answering endpoint; explicitly declare browsing.",
      "limitations": "Keep browse-enabled and closed-book results separate; narrower and often less discriminating than BrowseComp.",
      "priority": "P3",
      "maturity": "Historical factuality baseline",
      "sourceUrl": "https://github.com/openai/simple-evals",
      "catalogUrl": "https://hub.harborframework.com/datasets/openai/simpleqa",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Exa",
          "url": "https://exa.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Perplexity",
          "url": "https://www.perplexity.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Tavily",
          "url": "https://www.tavily.com/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "You.com",
          "url": "https://you.com/",
          "fit": "Research endpoint or composite system"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openai/simpleqa",
          "url": "https://hub.harborframework.com/datasets/openai/simpleqa",
          "tasks": 4326,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mteb-mmteb",
      "profileId": "B040",
      "name": "MTEB / MMTEB",
      "category": "Retrieval, embeddings and long context",
      "description": "Embedding quality across retrieval, similarity, classification and multilingual tasks.",
      "metrics": "Task-specific metrics; report retrieval nDCG separately from broad averages.",
      "execution": "Embedding/reranking endpoint through the selected MTEB task collection.",
      "limitations": "Vector databases should benchmark a declared embedding-plus-index pipeline; MTEB is not itself a database performance test.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/embeddings-benchmark/mteb",
      "catalogUrl": "https://huggingface.co/datasets/mteb/arguana",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Jina AI",
          "url": "https://jina.ai/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Voyage AI",
          "url": "https://www.voyageai.com/",
          "fit": "Direct component API + selected system adapters"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "mteb/arguana",
          "url": "https://huggingface.co/datasets/mteb/arguana",
          "tasks": null,
          "access": "Public",
          "review": "One MTEB/BEIR retrieval task; do not call this the complete MTEB suite."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "beir",
      "profileId": "B041",
      "name": "BEIR",
      "category": "Retrieval, embeddings and long context",
      "description": "Zero-shot retrieval across heterogeneous domains and corpora.",
      "metrics": "nDCG@10, recall and other declared retrieval metrics.",
      "execution": "Search/index pipeline with fixed corpus, chunking and embeddings.",
      "limitations": "For database comparisons, hold embeddings, retrieval configuration and recall targets fixed; report latency separately.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/beir-cellar/beir",
      "catalogUrl": "https://huggingface.co/BeIR",
      "companies": [
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Jina AI",
          "url": "https://jina.ai/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Pinecone",
          "url": "https://www.pinecone.io/",
          "fit": "Composite system"
        },
        {
          "name": "Qdrant",
          "url": "https://qdrant.tech/",
          "fit": "Composite system"
        },
        {
          "name": "Vectara",
          "url": "https://www.vectara.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Voyage AI",
          "url": "https://www.voyageai.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Weaviate",
          "url": "https://weaviate.io/",
          "fit": "Composite system"
        },
        {
          "name": "Zilliz",
          "url": "https://zilliz.com/",
          "fit": "Composite system"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "bright",
      "profileId": "B042",
      "name": "BRIGHT",
      "category": "Retrieval, embeddings and long context",
      "description": "Retrieval requiring reasoning beyond lexical or simple semantic matching.",
      "metrics": "nDCG@10 and domain breakdowns.",
      "execution": "Retrieval/reranker or agentic retrieval system.",
      "limitations": "Good complement to BEIR; label query expansion, reasoning model and additional compute.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://huggingface.co/datasets/mteb/BRIGHT",
      "catalogUrl": "https://huggingface.co/datasets/mteb/BRIGHT",
      "companies": [
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Exa",
          "url": "https://exa.ai/",
          "fit": "Research endpoint or composite system"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Jina AI",
          "url": "https://jina.ai/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Pinecone",
          "url": "https://www.pinecone.io/",
          "fit": "Composite system"
        },
        {
          "name": "Vectara",
          "url": "https://www.vectara.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Voyage AI",
          "url": "https://www.voyageai.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Weaviate",
          "url": "https://weaviate.io/",
          "fit": "Composite system"
        },
        {
          "name": "Qdrant",
          "url": "https://qdrant.tech/",
          "fit": "A configured retrieval pipeline must use the original corpus, exclusions and evaluator; disclose embeddings and any query reasoning."
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "mteb/BRIGHT",
          "url": "https://huggingface.co/datasets/mteb/BRIGHT",
          "tasks": null,
          "access": "Public",
          "review": "Selected reasoning-intensive retrieval suite."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "ruler",
      "profileId": "B043",
      "name": "RULER",
      "category": "Retrieval, embeddings and long context",
      "description": "Synthetic long-context retrieval, aggregation and reasoning at varying lengths.",
      "metrics": "Accuracy versus context length and task family.",
      "execution": "Long-context text model.",
      "limitations": "Context-window capacity is not demonstrated simply by accepting a long input; synthetic success also does not establish enterprise document quality.",
      "priority": "P2",
      "maturity": "Established diagnostic",
      "sourceUrl": "https://github.com/NVIDIA/RULER",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "longbench-v2",
      "profileId": "B044",
      "name": "LongBench v2",
      "category": "Retrieval, embeddings and long context",
      "description": "Long-document comprehension and reasoning across realistic scenarios.",
      "metrics": "Multiple-choice accuracy across length and difficulty slices.",
      "execution": "Long-context model or declared retrieval-assisted setting.",
      "limitations": "Report retrieval use and context truncation; avoid comparing different input access conditions.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/THUDM/LongBench",
      "catalogUrl": "https://huggingface.co/datasets/THUDM/LongBench-v2",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "bird",
      "profileId": "B045",
      "name": "BIRD",
      "category": "SQL, analytics and data engineering",
      "description": "Text-to-SQL over realistic databases and domain knowledge.",
      "metrics": "Execution accuracy; efficiency only under the declared official metric.",
      "execution": "SQL generation, database access and query runner.",
      "limitations": "SQL dialect and semantic-layer adaptations must be documented; report full split versus Mini-Dev.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://bird-bench.github.io/",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hex",
          "url": "https://hex.tech/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Julius AI",
          "url": "https://julius.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MotherDuck",
          "url": "https://motherduck.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sigma",
          "url": "https://www.sigmacomputing.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Snowflake",
          "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-analyst",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ThoughtSpot",
          "url": "https://www.thoughtspot.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "spider-2-0",
      "profileId": "B046",
      "name": "Spider 2.0",
      "category": "SQL, analytics and data engineering",
      "description": "Enterprise SQL/data workflows using substantial schemas, documentation and code.",
      "metrics": "Execution/task success on the selected Spider2 setting.",
      "execution": "Warehouse or local database environment; some tracks require external accounts.",
      "limitations": "Upstream reports a Snowflake evaluation-account suspension in August 2026; resolve access or select a supported local track before promising a run.",
      "priority": "P1",
      "maturity": "Established newer suite",
      "sourceUrl": "https://github.com/xlang-ai/Spider2",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hex",
          "url": "https://hex.tech/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Julius AI",
          "url": "https://julius.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MotherDuck",
          "url": "https://motherduck.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sigma",
          "url": "https://www.sigmacomputing.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Snowflake",
          "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-analyst",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ThoughtSpot",
          "url": "https://www.thoughtspot.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "dbt Labs",
          "url": "https://www.getdbt.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "dabstep",
      "profileId": "B047",
      "name": "DABstep",
      "category": "SQL, analytics and data engineering",
      "description": "450 multi-step data-analysis challenges derived from payments workloads.",
      "metrics": "Answer/task correctness under the benchmark evaluator.",
      "execution": "Data agent with Python/shell and benchmark files.",
      "limitations": "Different from pure text-to-SQL: requires reasoning over files, documentation and operational definitions.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://huggingface.co/datasets/adyen/DABstep",
      "catalogUrl": "https://hub.harborframework.com/datasets/adyen/dabstep",
      "companies": [
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hex",
          "url": "https://hex.tech/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Julius AI",
          "url": "https://julius.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MotherDuck",
          "url": "https://motherduck.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sigma",
          "url": "https://www.sigmacomputing.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Snowflake",
          "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-analyst",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ThoughtSpot",
          "url": "https://www.thoughtspot.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "adyen/dabstep",
          "url": "https://hub.harborframework.com/datasets/adyen/dabstep",
          "tasks": 450,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "dsbench",
      "profileId": "B048",
      "name": "DSBench",
      "category": "SQL, analytics and data engineering",
      "description": "Data analysis and data modeling tasks grounded in real data-science workflows.",
      "metrics": "Analysis-answer correctness and modeling metrics by task.",
      "execution": "Notebook/code agent with datasets and execution tools.",
      "limitations": "Distinguish analysis and modeling tracks; GPU/data dependencies may make costs uneven.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/LiqiangJing/DSBench",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hex",
          "url": "https://hex.tech/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Julius AI",
          "url": "https://julius.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Snowflake",
          "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-analyst",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "spreadsheetbench-verified",
      "profileId": "B049",
      "name": "SpreadsheetBench / Verified",
      "category": "SQL, analytics and data engineering",
      "description": "Real spreadsheet manipulation; full release has 912 questions and Verified has 400 instances.",
      "metrics": "Output workbook/cell correctness under the published evaluator.",
      "execution": "Agent capable of editing spreadsheet files.",
      "limitations": "A SQL-only product is not directly compatible. Declare full versus Verified and workbook tooling.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/RUCKBReasoning/SpreadsheetBench",
      "catalogUrl": "https://huggingface.co/datasets/KAKA22/SpreadsheetBench",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Julius AI",
          "url": "https://julius.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Microsoft Copilot",
          "url": "https://www.microsoft.com/en-us/microsoft-365-copilot",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Shortcut",
          "url": "https://shortcut.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hex",
          "url": "https://hex.tech/",
          "fit": "Conditional on a configured workflow accepting the required workbook and producing the native graded spreadsheet output."
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "ade-bench",
      "profileId": "B050",
      "name": "ADE-bench",
      "category": "SQL, analytics and data engineering",
      "description": "Sandboxed analytics engineering and data analyst tasks; Harbor snapshot has 48 tasks.",
      "metrics": "Expected warehouse/project outcome checks.",
      "execution": "dbt/data project and temporary data warehouse environment.",
      "limitations": "Track project setup and warehouse dialect; the public task sample is not a general BI leaderboard.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://github.com/dbt-labs/ade-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/dbt-labs/ade-bench",
      "companies": [
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hex",
          "url": "https://hex.tech/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MotherDuck",
          "url": "https://motherduck.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Snowflake",
          "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-analyst",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "dbt Labs",
          "url": "https://www.getdbt.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "dbt-labs/ade-bench",
          "url": "https://hub.harborframework.com/datasets/dbt-labs/ade-bench",
          "tasks": 48,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "data-eng-bench",
      "profileId": "B051",
      "name": "Data-Eng-Bench",
      "category": "SQL, analytics and data engineering",
      "description": "103 dbt tasks: analytics models, fixes, snapshots and incremental pipelines.",
      "metrics": "Hidden tests comparing materialized tables to reference outcomes.",
      "execution": "Harbor, dbt and DuckDB or the documented Snowflake setting.",
      "limitations": "Report all 103 tasks versus the 30-task fast subset; DuckDB/Snowflake configuration can change compatibility.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://hub.harborframework.com/datasets/snowflake-labs/data-eng-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/snowflake-labs/data-eng-bench",
      "companies": [
        {
          "name": "Augment Code",
          "url": "https://www.augmentcode.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Factory",
          "url": "https://docs.factory.ai/cli/getting-started/overview",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MotherDuck",
          "url": "https://motherduck.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Snowflake",
          "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-analyst",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "dbt Labs",
          "url": "https://www.getdbt.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "snowflake-labs/data-eng-bench",
          "url": "https://hub.harborframework.com/datasets/snowflake-labs/data-eng-bench",
          "tasks": 103,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "legalbench",
      "profileId": "B052",
      "name": "LegalBench",
      "category": "Legal AI",
      "description": "162 legal reasoning tasks spanning multiple legal capabilities.",
      "metrics": "Per-task classification/extraction metrics and macro summaries.",
      "execution": "Text model or appropriately configured legal assistant.",
      "limitations": "Task-level and source-data licenses differ. Static legal reasoning is not equivalent to completing a matter workflow.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/HazyResearch/legalbench",
      "catalogUrl": "https://huggingface.co/datasets/nguha/legalbench",
      "companies": [
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Ironclad",
          "url": "https://ironcladapp.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "LexisNexis",
          "url": "https://www.lexisnexis.com/en-us/products/lexis-plus-ai.page",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Luminance",
          "url": "https://www.luminance.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "legalbench-rag",
      "profileId": "B053",
      "name": "LegalBench-RAG",
      "category": "Legal AI",
      "description": "Retrieval of precise evidence spans for legal contract questions.",
      "metrics": "Character-level precision and recall.",
      "execution": "Retrieval pipeline over the provided legal corpus.",
      "limitations": "Measures retrieval, not the correctness of the generated legal analysis; preserve exact source offsets.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/zeroentropy-ai/legalbenchrag",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Ironclad",
          "url": "https://ironcladapp.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "LexisNexis",
          "url": "https://www.lexisnexis.com/en-us/products/lexis-plus-ai.page",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Luminance",
          "url": "https://www.luminance.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Voyage AI",
          "url": "https://www.voyageai.com/",
          "fit": "Direct component API + selected system adapters"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "cuad",
      "profileId": "B054",
      "name": "CUAD",
      "category": "Legal AI",
      "description": "Clause and key-information extraction from contracts.",
      "metrics": "Span extraction precision/recall/F1 or the published task metric.",
      "execution": "Contract extraction pipeline.",
      "limitations": "Use the original CUAD release and held-out split; this is an extraction benchmark rather than legal advice quality.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://www.atticusprojectai.org/cuad",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Ironclad",
          "url": "https://ironcladapp.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "LexisNexis",
          "url": "https://www.lexisnexis.com/en-us/products/lexis-plus-ai.page",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Luminance",
          "url": "https://www.luminance.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "contractnli",
      "profileId": "B055",
      "name": "ContractNLI",
      "category": "Legal AI",
      "description": "Entailment, contradiction and evidence identification in contracts.",
      "metrics": "NLI classification and evidence-span quality.",
      "execution": "Contract text plus hypothesis; model outputs labels and evidence.",
      "limitations": "A narrow contract-understanding diagnostic; do not infer negotiation capability from NLI accuracy.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://stanfordnlp.github.io/contract-nli/",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Ironclad",
          "url": "https://ironcladapp.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "LexisNexis",
          "url": "https://www.lexisnexis.com/en-us/products/lexis-plus-ai.page",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Luminance",
          "url": "https://www.luminance.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harvey-lab",
      "profileId": "B056",
      "name": "Harvey LAB",
      "category": "Legal AI",
      "description": "Long-horizon legal work with documents and per-task rubrics; Harbor snapshot has 1,251 tasks.",
      "metrics": "Rubric-criterion pass rate and all-criteria task pass rate.",
      "execution": "Legal agent with document/file tools and fixed rubric judge.",
      "limitations": "Publisher-created benchmark: disclose provenance. Main repository and third-party ports differ in task counts and expansions; freeze the chosen release.",
      "priority": "P1",
      "maturity": "Newer professional suite",
      "sourceUrl": "https://github.com/harveyai/harvey-labs",
      "catalogUrl": "https://hub.harborframework.com/datasets/harveyai/lab",
      "companies": [
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Ironclad",
          "url": "https://ironcladapp.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "LexisNexis",
          "url": "https://www.lexisnexis.com/en-us/products/lexis-plus-ai.page",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Luminance",
          "url": "https://www.luminance.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "harveyai/lab",
          "url": "https://hub.harborframework.com/datasets/harveyai/lab",
          "tasks": 1251,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "redlinebench",
      "profileId": "B057",
      "name": "RedlineBench",
      "category": "Legal AI",
      "description": "Multi-turn contract redlining and negotiation grounded in SaaS transactions.",
      "metrics": "Legal correctness, commercial alignment and negotiation-specific dimensions.",
      "execution": "Multi-turn contract editing/evaluation adapter.",
      "limitations": "Strong product alignment, but less established than LegalBench; keep negotiation setup and evaluator fixed.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/crosbylegal/RedlineBench",
      "catalogUrl": "https://huggingface.co/datasets/crosbylegal/RedlineBench",
      "companies": [
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Ironclad",
          "url": "https://ironcladapp.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "LexisNexis",
          "url": "https://www.lexisnexis.com/en-us/products/lexis-plus-ai.page",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Luminance",
          "url": "https://www.luminance.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "crosbylegal/RedlineBench",
          "url": "https://huggingface.co/datasets/crosbylegal/RedlineBench",
          "tasks": null,
          "access": "Public",
          "review": "Selected contract-negotiation specialist."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "lexam-lexam-hard",
      "profileId": "B058",
      "name": "LEXam / LEXam-hard",
      "category": "Legal AI",
      "description": "Law-exam reasoning focused on Swiss, EU and international law.",
      "metrics": "Question-format-specific answer scores.",
      "execution": "Legal QA model with declared tools.",
      "limitations": "Jurisdiction-sensitive; not a general U.S. law-firm readiness test. Hard is a filtered subset with a distinct selection process.",
      "priority": "P3",
      "maturity": "Newer regional specialist",
      "sourceUrl": "https://huggingface.co/datasets/LEXam-Benchmark/LEXam",
      "catalogUrl": "https://huggingface.co/datasets/joelniklaus/LEXam-hard",
      "companies": [
        {
          "name": "Harvey",
          "url": "https://www.harvey.ai/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Legora",
          "url": "https://legora.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "LexisNexis",
          "url": "https://www.lexisnexis.com/en-us/products/lexis-plus-ai.page",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "joelniklaus/LEXam-hard",
          "url": "https://huggingface.co/datasets/joelniklaus/LEXam-hard",
          "tasks": null,
          "access": "Public",
          "review": "Selected regional legal-reasoning extension; filtered hard subset."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "financebench",
      "profileId": "B059",
      "name": "FinanceBench",
      "category": "Financial analysis",
      "description": "Evidence-backed QA over company financial disclosures.",
      "metrics": "Answer correctness and supporting-evidence quality.",
      "execution": "Document retrieval and financial QA pipeline.",
      "limitations": "The public repository releases 150 examples, not all 10,231 described questions. Report the public-sample scope explicitly.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/patronus-ai/financebench",
      "catalogUrl": "https://huggingface.co/datasets/patronus-ai/financebench",
      "companies": [
        {
          "name": "AlphaSense",
          "url": "https://www.alpha-sense.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Thomson Reuters CoCounsel",
          "url": "https://legal.thomsonreuters.com/en/products/cocounsel",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "finqa-convfinqa",
      "profileId": "B060",
      "name": "FinQA / ConvFinQA",
      "category": "Financial analysis",
      "description": "Numerical reasoning over financial tables/text; conversational follow-ups in ConvFinQA.",
      "metrics": "Execution accuracy and program accuracy.",
      "execution": "Financial reasoning model with calculator/code tools where allowed.",
      "limitations": "Report whether external calculation tools are enabled. Historical annual-report data cannot establish live market-data coverage.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/czyssrs/FinQA",
      "catalogUrl": "",
      "companies": [
        {
          "name": "AlphaSense",
          "url": "https://www.alpha-sense.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Snowflake",
          "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-analyst",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "finben-pixiu",
      "profileId": "B061",
      "name": "FinBen / PIXIU",
      "category": "Financial analysis",
      "description": "Financial language tasks including extraction, classification and reasoning.",
      "metrics": "Per-task metrics rather than a single undifferentiated score.",
      "execution": "Task-specific financial model adapters.",
      "limitations": "Review individual dataset rights and the repository use terms; QA/classification performance does not establish investment performance.",
      "priority": "P3",
      "maturity": "Established specialist collection",
      "sourceUrl": "https://github.com/The-FinAI/PIXIU",
      "catalogUrl": "",
      "companies": [
        {
          "name": "AlphaSense",
          "url": "https://www.alpha-sense.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Bloomberg",
          "url": "https://arxiv.org/abs/2303.17564",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "omnidocbench",
      "profileId": "B062",
      "name": "OmniDocBench",
      "category": "Document parsing and extraction",
      "description": "PDF parsing of text, tables, formulas and reading order.",
      "metrics": "Component-specific edit distance, table/formula and ordering metrics.",
      "execution": "Document-to-structured-text/Markdown adapter.",
      "limitations": "Freeze dataset version, matching rules and render resolution; a single OCR string accuracy hides table and layout failures.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/opendatalab/OmniDocBench",
      "catalogUrl": "https://huggingface.co/datasets/opendatalab/OmniDocBench",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "olmocr-bench",
      "profileId": "B063",
      "name": "olmOCR-bench",
      "category": "Document parsing and extraction",
      "description": "PDF-to-Markdown fidelity; current card describes 1,403 PDFs and 7,010 unit tests.",
      "metrics": "Document property/unit-test pass rate.",
      "execution": "Document parser producing normalized Markdown.",
      "limitations": "The unit tests measure selected properties, not all possible extraction fields; use alongside a schema-extraction test.",
      "priority": "P1",
      "maturity": "Established newer suite",
      "sourceUrl": "https://huggingface.co/datasets/allenai/olmOCR-bench",
      "catalogUrl": "https://huggingface.co/datasets/allenai/olmOCR-bench",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "allenai/olmOCR-bench",
          "url": "https://huggingface.co/datasets/allenai/olmOCR-bench",
          "tasks": null,
          "access": "Public",
          "review": "Selected document parsing suite."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "parsebench",
      "profileId": "B064",
      "name": "ParseBench",
      "category": "Document parsing and extraction",
      "description": "Enterprise document parsing across tables, charts, faithfulness, formatting and grounding.",
      "metrics": "Capability-specific parsing metrics.",
      "execution": "Document parser with the required output representations.",
      "limitations": "LlamaIndex publishes the benchmark; disclose publisher interest and combine with independent suites.",
      "priority": "P1",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/llamaindex/ParseBench",
      "catalogUrl": "https://huggingface.co/datasets/llamaindex/ParseBench",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "llamaindex/ParseBench",
          "url": "https://huggingface.co/datasets/llamaindex/ParseBench",
          "tasks": null,
          "access": "Public",
          "review": "Selected enterprise document parsing suite; vendor-published."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "extractbench",
      "profileId": "B065",
      "name": "ExtractBench",
      "category": "Document parsing and extraction",
      "description": "Schema-driven document extraction with page/bounding-box evidence.",
      "metrics": "Value correctness, completeness, schema compliance and grounding.",
      "execution": "Structured extraction API returning schema-valid JSON and evidence.",
      "limitations": "An OCR-only endpoint needs an extraction layer; plain text output cannot satisfy all evidence requirements.",
      "priority": "P1",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/llamaindex/ExtractBench",
      "catalogUrl": "https://huggingface.co/datasets/llamaindex/ExtractBench",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "llamaindex/ExtractBench",
          "url": "https://huggingface.co/datasets/llamaindex/ExtractBench",
          "tasks": null,
          "access": "Public",
          "review": "Selected structured document extraction suite."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "ocrbench-ocrbench-v2",
      "profileId": "B066",
      "name": "OCRBench / OCRBench v2",
      "category": "Document parsing and extraction",
      "description": "Text recognition and document-oriented multimodal reasoning.",
      "metrics": "Task-specific correctness.",
      "execution": "Image/document question-answering model.",
      "limitations": "Separate original and v2; QA scores do not substitute for whole-document parsing quality.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/Yuliang-Liu/MultimodalOCR",
      "catalogUrl": "https://huggingface.co/datasets/ling99/OCRBench_v2",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "real5-omnidocbench-mdpbench",
      "profileId": "B067",
      "name": "Real5-OmniDocBench / MDPBench",
      "category": "Document parsing and extraction",
      "description": "Physical capture distortions and multilingual document parsing.",
      "metrics": "Parsing scores by capture condition, script and language.",
      "execution": "Document parser accepting scanned/photographed inputs.",
      "limitations": "Use robustness slices alongside a clean-document baseline; do not call all official-tagged HF entries equally established.",
      "priority": "P2",
      "maturity": "Newer robustness extensions",
      "sourceUrl": "https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench",
      "catalogUrl": "https://huggingface.co/datasets/Delores-Lin/MDPBench",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "Delores-Lin/MDPBench",
          "url": "https://huggingface.co/datasets/Delores-Lin/MDPBench",
          "tasks": null,
          "access": "Public",
          "review": "Selected multilingual document robustness extension."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "seed-tts-eval",
      "profileId": "B068",
      "name": "Seed-TTS-eval",
      "category": "Speech synthesis and voice cloning",
      "description": "Zero-shot TTS/voice conversion in English and Mandarin using reference audio.",
      "metrics": "WER and speaker-similarity cosine score under the official ASR/embedding stack.",
      "execution": "Reference-conditioned TTS API; generate WAV/audio for the prescribed text.",
      "limitations": "Full protocol requires compatible reference clips and permitted cloning access. Deepgram/WellSaid stock voices cannot be assigned an equivalent speaker-cloning score; stock-voice WER is a separate adaptation.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/BytedanceSpeech/seed-tts-eval",
      "catalogUrl": "https://huggingface.co/datasets/hhqx/seedtts_testset",
      "companies": [
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Fish Audio",
          "url": "https://docs.fish.audio/api-reference/sdk/python/resources",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "HeyGen",
          "url": "https://www.heygen.com/tool/ai-voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Murf AI",
          "url": "https://murf.ai/api/docs/api-reference/voice-cloning/create",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Resemble AI",
          "url": "https://www.resemble.ai/",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Speechify",
          "url": "https://docs.speechify.ai/tts/guides/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Synthesia",
          "url": "https://docs.synthesia.io/docs/custom-voices",
          "fit": "TTS API; full cloning protocol conditional"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "open-tts-leaderboard",
      "profileId": "B069",
      "name": "Open TTS Leaderboard",
      "category": "Speech synthesis and voice cloning",
      "description": "Listener preferences for generated speech samples.",
      "metrics": "Human preference/vote outcomes under the leaderboard protocol.",
      "execution": "TTS audio samples and the specified listening comparison.",
      "limitations": "A human-evaluation surface, not a deterministic offline dataset; useful complement to Seed intelligibility/similarity metrics.",
      "priority": "P2",
      "maturity": "Newer human-preference complement",
      "sourceUrl": "https://huggingface.co/spaces/hf-audio/open_tts_leaderboard",
      "catalogUrl": "https://huggingface.co/spaces/hf-audio/open_tts_leaderboard",
      "companies": [
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Fish Audio",
          "url": "https://docs.fish.audio/api-reference/sdk/python/resources",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Hume AI",
          "url": "https://www.hume.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Murf AI",
          "url": "https://murf.ai/api/docs/api-reference/voice-cloning/create",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Resemble AI",
          "url": "https://www.resemble.ai/",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Speechify",
          "url": "https://docs.speechify.ai/tts/guides/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "WellSaid",
          "url": "https://help.wellsaid.io/hc/en-us/articles/39112914360467-List-of-WellSaid-Voices",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "open-asr-leaderboard-esb",
      "profileId": "B070",
      "name": "Open ASR Leaderboard / ESB",
      "category": "Speech recognition",
      "description": "Speech transcription across multiple domains, with multilingual and long-form tracks.",
      "metrics": "Normalized WER plus speed/throughput under a declared protocol.",
      "execution": "Speech-to-text API or local model.",
      "limitations": "The suite is not a single uniform corpus license. Normalize transcripts identically and separate long-form, multilingual and English tracks.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/blog/open-asr-leaderboard",
      "catalogUrl": "https://huggingface.co/datasets/hf-audio/open-asr-leaderboard",
      "companies": [
        {
          "name": "AssemblyAI",
          "url": "https://www.assemblyai.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Gladia",
          "url": "https://www.gladia.io/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rev",
          "url": "https://www.rev.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Soniox",
          "url": "https://soniox.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Speechmatics",
          "url": "https://www.speechmatics.com/",
          "fit": "Direct ASR API"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "hf-audio/open-asr-leaderboard",
          "url": "https://huggingface.co/datasets/hf-audio/open-asr-leaderboard",
          "tasks": null,
          "access": "Public",
          "review": "Selected ASR evaluation collection; check constituent corpus terms."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "librispeech-test-clean-test-other",
      "profileId": "B071",
      "name": "LibriSpeech test-clean / test-other",
      "category": "Speech recognition",
      "description": "Read English speech transcription under cleaner and harder conditions.",
      "metrics": "WER per split.",
      "execution": "Speech recognition endpoint.",
      "limitations": "Useful baseline, but clean read speech is insufficient for contact-center or meeting claims.",
      "priority": "P2",
      "maturity": "Established historical baseline",
      "sourceUrl": "https://www.openslr.org/12",
      "catalogUrl": "",
      "companies": [
        {
          "name": "AssemblyAI",
          "url": "https://www.assemblyai.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Gladia",
          "url": "https://www.gladia.io/",
          "fit": "Direct ASR API"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rev",
          "url": "https://www.rev.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Soniox",
          "url": "https://soniox.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Speechmatics",
          "url": "https://www.speechmatics.com/",
          "fit": "Direct ASR API"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "fleurs",
      "profileId": "B072",
      "name": "FLEURS",
      "category": "Speech recognition",
      "description": "Multilingual speech across 102 languages.",
      "metrics": "WER/CER by language under the official normalization.",
      "execution": "ASR or speech system supporting the evaluated languages.",
      "limitations": "Report unsupported languages and macro weighting; a global average can hide weak low-resource-language performance.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/google/fleurs",
      "catalogUrl": "https://huggingface.co/datasets/google/fleurs",
      "companies": [
        {
          "name": "AssemblyAI",
          "url": "https://www.assemblyai.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Gladia",
          "url": "https://www.gladia.io/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rev",
          "url": "https://www.rev.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Soniox",
          "url": "https://soniox.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Speechmatics",
          "url": "https://www.speechmatics.com/",
          "fit": "Direct ASR API"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "common-voice",
      "profileId": "B073",
      "name": "Common Voice",
      "category": "Speech recognition",
      "description": "Crowd-sourced speech spanning languages, accents and speakers.",
      "metrics": "WER/CER on a frozen release and held-out split.",
      "execution": "Speech recognizer plus exact release and language selection.",
      "limitations": "A large evolving dataset, not one fixed leaderboard protocol; freeze release, split and licenses.",
      "priority": "P2",
      "maturity": "Established corpus with evaluation splits",
      "sourceUrl": "https://commonvoice.mozilla.org/en/datasets",
      "catalogUrl": "",
      "companies": [
        {
          "name": "AssemblyAI",
          "url": "https://www.assemblyai.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Gladia",
          "url": "https://www.gladia.io/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rev",
          "url": "https://www.rev.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Soniox",
          "url": "https://soniox.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Speechmatics",
          "url": "https://www.speechmatics.com/",
          "fit": "Direct ASR API"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "earnings-22",
      "profileId": "B074",
      "name": "Earnings-22",
      "category": "Speech recognition",
      "description": "Long-form earnings-call speech with diverse accents.",
      "metrics": "Long-form WER under the selected segmentation and normalization.",
      "execution": "ASR system handling long audio.",
      "limitations": "Transcription quality does not establish diarization, financial-summary accuracy or speaker attribution.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/revdotcom/speech-datasets/tree/main/earnings22",
      "catalogUrl": "",
      "companies": [
        {
          "name": "AlphaSense",
          "url": "https://www.alpha-sense.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "AssemblyAI",
          "url": "https://www.assemblyai.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Fireflies",
          "url": "https://fireflies.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Gladia",
          "url": "https://www.gladia.io/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Otter",
          "url": "https://otter.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rev",
          "url": "https://www.rev.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Soniox",
          "url": "https://soniox.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Speechmatics",
          "url": "https://www.speechmatics.com/",
          "fit": "Direct ASR API"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "ami-meeting-corpus",
      "profileId": "B075",
      "name": "AMI meeting corpus",
      "category": "Speech recognition",
      "description": "Meeting speech with multiple speakers and interaction.",
      "metrics": "WER plus separately measured diarization error where configured.",
      "execution": "Meeting transcription and speaker-diarization pipeline.",
      "limitations": "ASR and diarization are different objectives; specify microphone channel, overlap treatment and scoring collar.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://groups.inf.ed.ac.uk/ami/corpus/",
      "catalogUrl": "",
      "companies": [
        {
          "name": "AssemblyAI",
          "url": "https://www.assemblyai.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Fireflies",
          "url": "https://fireflies.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Gladia",
          "url": "https://www.gladia.io/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Otter",
          "url": "https://otter.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rev",
          "url": "https://www.rev.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Speechmatics",
          "url": "https://www.speechmatics.com/",
          "fit": "Direct ASR API"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "vaani-benchmark-v1-0",
      "profileId": "B076",
      "name": "Vaani Benchmark v1.0",
      "category": "Speech recognition",
      "description": "Speech from diverse Indian districts and speakers with multiple reference transcripts.",
      "metrics": "Benchmark-specific multilingual transcription scores.",
      "execution": "ASR with supported Indian languages.",
      "limitations": "Gated access and language-specific normalization; supplement rather than replace global speech coverage.",
      "priority": "P3",
      "maturity": "Newer regional specialist",
      "sourceUrl": "https://huggingface.co/datasets/ARTPARK-IISc/Vaani-Benchmark-V1.0",
      "catalogUrl": "https://huggingface.co/datasets/ARTPARK-IISc/Vaani-Benchmark-V1.0",
      "companies": [
        {
          "name": "AssemblyAI",
          "url": "https://www.assemblyai.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Gladia",
          "url": "https://www.gladia.io/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rev",
          "url": "https://www.rev.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Soniox",
          "url": "https://soniox.com/",
          "fit": "Direct ASR API"
        },
        {
          "name": "Speechmatics",
          "url": "https://www.speechmatics.com/",
          "fit": "Direct ASR API"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "ARTPARK-IISc/Vaani-Benchmark-V1.0",
          "url": "https://huggingface.co/datasets/ARTPARK-IISc/Vaani-Benchmark-V1.0",
          "tasks": null,
          "access": "Gated",
          "review": "Selected Indian-language ASR specialist; gated."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "voice-full-duplex-voice",
      "profileId": "B077",
      "name": "τ-Voice / τ³ full-duplex voice",
      "category": "Conversational voice and audio understanding",
      "description": "Spoken service conversations with tools, interruptions and simultaneous audio.",
      "metrics": "Task reward plus separately logged latency, turn-taking and interruption recovery.",
      "execution": "Native real-time audio adapter and benchmark tool/user simulation.",
      "limitations": "This measures the complete voice agent, not TTS alone. The upstream supported-provider list is narrower than the prospect list; adapters are needed.",
      "priority": "P1",
      "maturity": "Newer high-fit agent suite",
      "sourceUrl": "https://arxiv.org/abs/2603.13686",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Bland AI",
          "url": "https://www.bland.ai/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Cresta",
          "url": "https://cresta.com/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Decagon",
          "url": "https://decagon.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Hume AI",
          "url": "https://www.hume.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Parloa",
          "url": "https://www.parloa.com/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "PolyAI",
          "url": "https://poly.ai/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Retell AI",
          "url": "https://www.retellai.com/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Sierra",
          "url": "https://sierra.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Vapi",
          "url": "https://vapi.ai/",
          "fit": "Full voice-stack adapter"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "voicebench",
      "profileId": "B078",
      "name": "VoiceBench",
      "category": "Conversational voice and audio understanding",
      "description": "Speech-input instruction following, reasoning and robustness to speaker/environment variation.",
      "metrics": "Subtask-specific accuracy / judge scores.",
      "execution": "Voice assistant with audio input and benchmark-compatible answers.",
      "limitations": "Not a replacement for telephone-network, full-duplex or transaction-success testing.",
      "priority": "P1",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/MatthewCYM/VoiceBench",
      "catalogUrl": "https://huggingface.co/datasets/hlt-lab/voicebench",
      "companies": [
        {
          "name": "Bland AI",
          "url": "https://www.bland.ai/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Cartesia",
          "url": "https://docs.cartesia.ai/api-reference/voices/clone",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Cresta",
          "url": "https://cresta.com/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Deepgram",
          "url": "https://developers.deepgram.com/docs/tts-models",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ElevenLabs",
          "url": "https://elevenlabs.io/docs/eleven-api/concepts/voice-cloning",
          "fit": "TTS API; full cloning protocol conditional"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hume AI",
          "url": "https://www.hume.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Parloa",
          "url": "https://www.parloa.com/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "PolyAI",
          "url": "https://poly.ai/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Retell AI",
          "url": "https://www.retellai.com/",
          "fit": "Full voice-stack adapter"
        },
        {
          "name": "Vapi",
          "url": "https://vapi.ai/",
          "fit": "Full voice-stack adapter"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mmau",
      "profileId": "B079",
      "name": "MMAU",
      "category": "Conversational voice and audio understanding",
      "description": "Audio understanding/reasoning over speech, sound and music.",
      "metrics": "Question-answer accuracy by audio category.",
      "execution": "Audio-language model receiving the original clips.",
      "limitations": "Harbor's 1,000-task adapter uses Qwen2-Audio transcripts for text models. That port does not measure native audio perception and must be labeled separately.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/Sakshi113/mmau",
      "catalogUrl": "https://huggingface.co/datasets/gamma-lab-umd/MMAU-test-mini",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hume AI",
          "url": "https://www.hume.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Twelve Labs",
          "url": "https://www.twelvelabs.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "geneval",
      "profileId": "B080",
      "name": "GenEval",
      "category": "Image and video generation",
      "description": "Text-to-image compositional correctness: objects, counts, colors and spatial relations.",
      "metrics": "Detector-based compositional success.",
      "execution": "Image generation API with fixed prompts, seeds and sample counts.",
      "limitations": "Useful for prompt adherence, not a complete aesthetic/typography score; detector limitations can affect results.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/djghosh13/geneval",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Adobe",
          "url": "https://www.adobe.com/products/firefly.html",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Black Forest Labs",
          "url": "https://bfl.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Canva",
          "url": "https://www.canva.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Ideogram",
          "url": "https://ideogram.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Midjourney",
          "url": "https://www.midjourney.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Stability AI",
          "url": "https://stability.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "dpg-bench",
      "profileId": "B081",
      "name": "DPG-Bench",
      "category": "Image and video generation",
      "description": "Dense-prompt image generation and fine-grained instruction following.",
      "metrics": "Question-based semantic alignment metrics.",
      "execution": "Image generator and fixed vision-language evaluator.",
      "limitations": "Judge/version sensitivity; report separately from human visual preference.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/TencentQQGYLab/ELLA",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Adobe",
          "url": "https://www.adobe.com/products/firefly.html",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Black Forest Labs",
          "url": "https://bfl.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Canva",
          "url": "https://www.canva.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Ideogram",
          "url": "https://ideogram.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Midjourney",
          "url": "https://www.midjourney.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Stability AI",
          "url": "https://stability.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "t2i-compbench-t2i-compbench",
      "profileId": "B082",
      "name": "T2I-CompBench / T2I-CompBench++",
      "category": "Image and video generation",
      "description": "Attribute binding, object relations and complex compositions.",
      "metrics": "Subcategory compositional metrics.",
      "execution": "Text-to-image generator and published evaluators.",
      "limitations": "Freeze benchmark version and evaluator models; avoid presenting a single component metric as overall image quality.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/Karine-Huang/T2I-CompBench",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Adobe",
          "url": "https://www.adobe.com/products/firefly.html",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Black Forest Labs",
          "url": "https://bfl.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Canva",
          "url": "https://www.canva.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Ideogram",
          "url": "https://ideogram.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Midjourney",
          "url": "https://www.midjourney.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Stability AI",
          "url": "https://stability.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "vbench-vbench-2-0",
      "profileId": "B083",
      "name": "VBench / VBench 2.0",
      "category": "Image and video generation",
      "description": "Generated video quality and consistency; 2.0 adds intrinsic faithfulness capabilities.",
      "metrics": "Dimension-level video scores.",
      "execution": "Video model with matched prompts, duration and resolution.",
      "limitations": "Compare like-for-like generation settings. General video quality does not establish avatar identity, lip sync or dubbing quality.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/Vchitect/VBench",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Adobe",
          "url": "https://www.adobe.com/products/firefly.html",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Kling",
          "url": "https://klingai.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Luma",
          "url": "https://lumalabs.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Pika",
          "url": "https://pika.art/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Runway",
          "url": "https://runwayml.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "wbench",
      "profileId": "B084",
      "name": "WBench",
      "category": "Image and video generation",
      "description": "Multi-turn interactive video-world generation.",
      "metrics": "Published dimensions for quality, setting adherence and interaction behavior.",
      "execution": "World-model API that accepts sequential actions/prompts.",
      "limitations": "Only relevant when the product exposes interactive world-model behavior; ordinary one-shot video APIs may be incompatible.",
      "priority": "P3",
      "maturity": "Newer world-model specialist",
      "sourceUrl": "https://huggingface.co/datasets/meituan-longcat/WBench",
      "catalogUrl": "https://huggingface.co/datasets/meituan-longcat/WBench",
      "companies": [
        {
          "name": "Black Forest Labs",
          "url": "https://bfl.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Luma",
          "url": "https://lumalabs.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Runway",
          "url": "https://runwayml.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "meituan-longcat/WBench",
          "url": "https://huggingface.co/datasets/meituan-longcat/WBench",
          "tasks": null,
          "access": "Public",
          "review": "Selected interactive video-world specialist, not ordinary one-shot video generation."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mmmu-pro",
      "profileId": "B085",
      "name": "MMMU-Pro",
      "category": "Multimodal understanding",
      "description": "Multidisciplinary reasoning over images and expert-level questions.",
      "metrics": "Answer accuracy by domain and setting.",
      "execution": "Vision-language model.",
      "limitations": "Use the corrected current labels and distinguish visual/standard settings. It does not test image generation.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/MMMU/MMMU_Pro",
      "catalogUrl": "https://huggingface.co/datasets/MMMU/MMMU_Pro",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Twelve Labs",
          "url": "https://www.twelvelabs.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "MMMU/MMMU_Pro",
          "url": "https://huggingface.co/datasets/MMMU/MMMU_Pro",
          "tasks": null,
          "access": "Public",
          "review": "Selected multimodal-understanding benchmark."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "video-mme-video-mme-v2",
      "profileId": "B086",
      "name": "Video-MME / Video-MME v2",
      "category": "Multimodal understanding",
      "description": "Video understanding, temporal reasoning and cross-video-length performance.",
      "metrics": "Question accuracy by task, duration and input setting.",
      "execution": "Video model with declared frame sampling, audio and subtitle access.",
      "limitations": "Understanding is distinct from generation. Changing subtitle/audio access changes the test.",
      "priority": "P2",
      "maturity": "Established family; newer v2",
      "sourceUrl": "https://huggingface.co/datasets/MME-Benchmarks/Video-MME-v2",
      "catalogUrl": "https://huggingface.co/datasets/MME-Benchmarks/Video-MME-v2",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Runway",
          "url": "https://runwayml.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Twelve Labs",
          "url": "https://www.twelvelabs.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "MME-Benchmarks/Video-MME-v2",
          "url": "https://huggingface.co/datasets/MME-Benchmarks/Video-MME-v2",
          "tasks": null,
          "access": "Public",
          "review": "Selected current video-understanding extension."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "agentdojo",
      "profileId": "B087",
      "name": "AgentDojo",
      "category": "Agent safety and security",
      "description": "Indirect prompt-injection attacks against useful tool workflows.",
      "metrics": "Attack success and benign task utility, reported together.",
      "execution": "Agent + defensive product in the same benchmark workflow.",
      "limitations": "Refusing every task is not a useful defense. Compare attack resilience at preserved task utility.",
      "priority": "P1",
      "maturity": "Established",
      "sourceUrl": "https://github.com/ethz-spylab/agentdojo",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Browser Use",
          "url": "https://browser-use.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "HiddenLayer",
          "url": "https://hiddenlayer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lakera",
          "url": "https://www.lakera.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Prompt Security",
          "url": "https://www.prompt.security/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Protect AI",
          "url": "https://protectai.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "injecagent",
      "profileId": "B088",
      "name": "InjecAgent",
      "category": "Agent safety and security",
      "description": "1,054 indirect-injection cases involving user and attacker tools.",
      "metrics": "Attack success rate under the prescribed scenarios.",
      "execution": "Tool-integrated model/agent adapter.",
      "limitations": "Static test cases complement interactive evaluation; do not equate them with all production injection routes.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/uiuc-kang-lab/InjecAgent",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Composio",
          "url": "https://composio.dev/",
          "fit": "Composite system"
        },
        {
          "name": "HiddenLayer",
          "url": "https://hiddenlayer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lakera",
          "url": "https://www.lakera.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Prompt Security",
          "url": "https://www.prompt.security/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Protect AI",
          "url": "https://protectai.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zapier",
          "url": "https://zapier.com/agents",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "agentharm",
      "profileId": "B089",
      "name": "AgentHarm",
      "category": "Agent safety and security",
      "description": "Harmful multi-step agent requests and refusal/behavioral robustness.",
      "metrics": "Benchmark harm/completion scores and benign counterpart performance.",
      "execution": "Agent using the benchmark tool harness.",
      "limitations": "This is a harmful-request evaluation, not specifically a prompt-injection benchmark; use the published release subset.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/UKGovernmentBEIS/Inspect_evals/tree/main/src/inspect_evals/agentharm",
      "catalogUrl": "https://huggingface.co/datasets/ai-safety-institute/AgentHarm",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "HiddenLayer",
          "url": "https://hiddenlayer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lakera",
          "url": "https://www.lakera.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harmbench-strongreject",
      "profileId": "B090",
      "name": "HarmBench / StrongREJECT",
      "category": "Agent safety and security",
      "description": "Safety robustness against harmful prompts and jailbreak attempts.",
      "metrics": "Attack success / refusal-quality metrics from each suite.",
      "execution": "Text model plus fixed attacks and graders.",
      "limitations": "Report false positives and helpfulness alongside safety; Harbor StrongREJECT is a 150-task package, not evidence of full-suite equivalence.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/centerforaisafety/HarmBench",
      "catalogUrl": "https://hub.harborframework.com/datasets/strongreject/strongreject",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "HiddenLayer",
          "url": "https://hiddenlayer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Lakera",
          "url": "https://www.lakera.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Protect AI",
          "url": "https://protectai.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "strongreject/strongreject",
          "url": "https://hub.harborframework.com/datasets/strongreject/strongreject",
          "tasks": 150,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "cybergym",
      "profileId": "B091",
      "name": "CyberGym",
      "category": "Agent safety and security",
      "description": "Real-world vulnerability analysis in reproducible software environments.",
      "metrics": "Verified task/vulnerability reproduction success.",
      "execution": "Isolated security agent and benchmark binaries.",
      "limitations": "Fit is limited to products with a vulnerability-analysis agent; an AI firewall alone cannot perform this benchmark.",
      "priority": "P3",
      "maturity": "Established newer specialist",
      "sourceUrl": "https://github.com/sunblaze-ucb/cybergym",
      "catalogUrl": "https://huggingface.co/datasets/sunblaze-ucb/cybergym",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "GitHub Copilot",
          "url": "https://github.com/features/copilot",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Protect AI",
          "url": "https://protectai.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sentry",
          "url": "https://sentry.io/product/seer/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "o11y-bench",
      "profileId": "B092",
      "name": "o11y-bench",
      "category": "DevOps and observability",
      "description": "63 tasks involving observability logs, metrics, traces, dashboards and incidents.",
      "metrics": "Task-specific verified outcomes.",
      "execution": "Agent connected to the benchmark observability environment.",
      "limitations": "Publisher-specific environment; qualify integration requirements and avoid treating incident-chat quality as actual remediation.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://hub.harborframework.com/datasets/grafana/o11y-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/grafana/o11y-bench",
      "companies": [
        {
          "name": "Datadog",
          "url": "https://www.datadoghq.com/product/bits-ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Grafana",
          "url": "https://github.com/grafana/o11y-bench",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "PagerDuty",
          "url": "https://www.pagerduty.com/platform/aiops/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Resolve AI",
          "url": "https://resolve.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sentry",
          "url": "https://sentry.io/product/seer/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "ServiceNow",
          "url": "https://www.servicenow.com/products/ai-agents.html",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "grafana/o11y-bench",
          "url": "https://hub.harborframework.com/datasets/grafana/o11y-bench",
          "tasks": 63,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "itsmbench",
      "profileId": "B093",
      "name": "ITSMBench",
      "category": "DevOps and observability",
      "description": "53 multi-turn IT service tasks with policies and stateful business records.",
      "metrics": "Binary final database-state correctness.",
      "execution": "Benchmark tenant, tool adapter and simulated colleague.",
      "limitations": "Requires benchmark tool semantics and identity/policy fidelity; useful niche fit rather than established broad fame.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://hub.harborframework.com/datasets/vibrantlabsai/itsm-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/vibrantlabsai/itsm-bench",
      "companies": [
        {
          "name": "Automation Anywhere",
          "url": "https://www.automationanywhere.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Salesforce",
          "url": "https://www.salesforce.com/agentforce/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "ServiceNow",
          "url": "https://www.servicenow.com/products/ai-agents.html",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "UiPath",
          "url": "https://www.uipath.com/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "vibrantlabsai/itsm-bench",
          "url": "https://hub.harborframework.com/datasets/vibrantlabsai/itsm-bench",
          "tasks": 53,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "healthbench",
      "profileId": "B094",
      "name": "HealthBench",
      "category": "Healthcare AI",
      "description": "Health conversations graded against clinician-written criteria.",
      "metrics": "Rubric scores under the published model-based grader.",
      "execution": "Health conversational model plus reference evaluator.",
      "limitations": "For ambient scribes this is only an adjacent reasoning check; it does not evaluate transcription, note fidelity or EHR integration.",
      "priority": "P3",
      "maturity": "Established newer specialist",
      "sourceUrl": "https://openai.com/index/healthbench/",
      "catalogUrl": "https://huggingface.co/datasets/openai/healthbench",
      "companies": [
        {
          "name": "Abridge",
          "url": "https://www.abridge.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Ambience",
          "url": "https://www.ambiencehealthcare.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hippocratic AI",
          "url": "https://www.hippocraticai.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Nabla",
          "url": "https://www.nabla.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Suki",
          "url": "https://www.suki.ai/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "medqa",
      "profileId": "B095",
      "name": "MedQA",
      "category": "Healthcare AI",
      "description": "Medical examination question answering.",
      "metrics": "Multiple-choice accuracy.",
      "execution": "Medical QA model.",
      "limitations": "Exam performance is not workflow performance or clinical validation; low priority for ambient-documentation products.",
      "priority": "P3",
      "maturity": "Established historical baseline",
      "sourceUrl": "https://github.com/jind11/MedQA",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hippocratic AI",
          "url": "https://www.hippocraticai.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "medagentbench",
      "profileId": "B096",
      "name": "MedAgentBench",
      "category": "Healthcare AI",
      "description": "300 tasks requiring action in a virtual FHIR-compatible EHR.",
      "metrics": "Task completion against environment state.",
      "execution": "FHIR/EHR action adapter and benchmark environment.",
      "limitations": "Only relevant to an action-capable product; a scribe without EHR tool access cannot complete the full suite.",
      "priority": "P3",
      "maturity": "Established newer specialist",
      "sourceUrl": "https://stanfordmlgroup.github.io/projects/medagentbench/",
      "catalogUrl": "https://hub.harborframework.com/datasets/stanford/medagentbench",
      "companies": [
        {
          "name": "Abridge",
          "url": "https://www.abridge.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Ambience",
          "url": "https://www.ambiencehealthcare.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hippocratic AI",
          "url": "https://www.hippocraticai.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Nabla",
          "url": "https://www.nabla.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Suki",
          "url": "https://www.suki.ai/",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "stanford/medagentbench",
          "url": "https://hub.harborframework.com/datasets/stanford/medagentbench",
          "tasks": 300,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "chi-bench",
      "profileId": "B097",
      "name": "CHI-Bench",
      "category": "Healthcare AI",
      "description": "Prior authorization, utilization management and care-management workflows.",
      "metrics": "State/policy-aware workflow outcome metrics.",
      "execution": "Healthcare-app tools and approved access to the gated operations handbook.",
      "limitations": "Harbor includes 78 single-agent tasks; 23 provider-payer E2E arena tasks use a different two-agent harness. Do not combine these task counts.",
      "priority": "P3",
      "maturity": "Newer specialist",
      "sourceUrl": "https://hub.harborframework.com/datasets/actava-ai/chi-bench",
      "catalogUrl": "https://huggingface.co/datasets/actava/chi-bench",
      "companies": [
        {
          "name": "Ambience",
          "url": "https://www.ambiencehealthcare.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Hippocratic AI",
          "url": "https://www.hippocraticai.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "Nabla",
          "url": "https://www.nabla.com/",
          "fit": "Product-specific adapter / qualified access"
        },
        {
          "name": "ServiceNow",
          "url": "https://www.servicenow.com/products/ai-agents.html",
          "fit": "Product-specific adapter / qualified access"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "actava/chi-bench",
          "url": "https://huggingface.co/datasets/actava/chi-bench",
          "tasks": null,
          "access": "Public",
          "review": "Selected healthcare-workflow specialist; runtime needs a separately gated handbook."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mle-bench",
      "profileId": "B098",
      "name": "MLE-bench",
      "category": "Scientific and ML agents",
      "description": "Machine-learning engineering on Kaggle-derived competitions.",
      "metrics": "Competition scores / medal-level attainment under the official protocol.",
      "execution": "Dataset downloads, compute budget and training/code agent.",
      "limitations": "Heavy compute and competition-specific terms; the repository currently pauses new leaderboard submissions while revising fairness checks.",
      "priority": "P3",
      "maturity": "Established",
      "sourceUrl": "https://github.com/openai/mle-bench",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sakana AI",
          "url": "https://sakana.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "re-bench",
      "profileId": "B099",
      "name": "RE-Bench",
      "category": "Scientific and ML agents",
      "description": "Autonomous AI research and engineering tasks with varying time budgets.",
      "metrics": "Task-specific improvement normalized by the published methodology.",
      "execution": "Research coding environment and fixed compute/time limits.",
      "limitations": "Small expert-designed suite; repeated exposure and resource differences can dominate comparisons.",
      "priority": "P3",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/METR/RE-Bench",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sakana AI",
          "url": "https://sakana.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "scienceagentbench-verified",
      "profileId": "B100",
      "name": "ScienceAgentBench Verified",
      "category": "Scientific and ML agents",
      "description": "Data-driven scientific coding tasks; original suite has 102 instances.",
      "metrics": "Executable task/output verification.",
      "execution": "Scientific code agent with datasets and declared GPU requirements.",
      "limitations": "Use the verified update released in April 2026. Search-only products need code execution before they fit the full protocol.",
      "priority": "P3",
      "maturity": "Established specialist",
      "sourceUrl": "https://github.com/OSU-NLP-Group/ScienceAgentBench",
      "catalogUrl": "https://huggingface.co/datasets/osunlp/ScienceAgentBench",
      "companies": [
        {
          "name": "Elicit",
          "url": "https://elicit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "FutureHouse",
          "url": "https://www.futurehouse.org/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sakana AI",
          "url": "https://sakana.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "lab-bench-futurehouse",
      "profileId": "B101",
      "name": "LAB-Bench (FutureHouse)",
      "category": "Scientific and ML agents",
      "description": "Biology literature, database, figure and procedural-reasoning questions.",
      "metrics": "Subtask answer accuracy.",
      "execution": "Science assistant with track-appropriate tools.",
      "limitations": "Different from Harvey LAB. The Harbor 181-task port is not the full upstream multi-subtask corpus.",
      "priority": "P3",
      "maturity": "Established specialist",
      "sourceUrl": "https://huggingface.co/datasets/futurehouse/lab-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/futurehouse/labbench",
      "companies": [
        {
          "name": "Consensus",
          "url": "https://consensus.app/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Elicit",
          "url": "https://elicit.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "FutureHouse",
          "url": "https://www.futurehouse.org/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "futurehouse/labbench",
          "url": "https://hub.harborframework.com/datasets/futurehouse/labbench",
          "tasks": 181,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "bixbench",
      "profileId": "B102",
      "name": "BixBench",
      "category": "Scientific and ML agents",
      "description": "Bioinformatics analysis using data capsules.",
      "metrics": "Analysis-answer correctness under the chosen answer setting.",
      "execution": "Python/R-capable scientific agent and per-task data.",
      "limitations": "Use corrected post-September-2025 tasks and report open-answer versus multiple-choice settings.",
      "priority": "P3",
      "maturity": "Established newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/futurehouse/BixBench",
      "catalogUrl": "https://hub.harborframework.com/datasets/futurehouse/bixbench",
      "companies": [
        {
          "name": "Databricks",
          "url": "https://www.databricks.com/product/business-intelligence/ai-bi-genie",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "FutureHouse",
          "url": "https://www.futurehouse.org/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sakana AI",
          "url": "https://sakana.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "futurehouse/bixbench",
          "url": "https://hub.harborframework.com/datasets/futurehouse/bixbench",
          "tasks": 205,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "terminal-bench-science",
      "profileId": "B103",
      "name": "Terminal-Bench-Science",
      "category": "Scientific and ML agents",
      "description": "Computational research workflows across scientific fields; Harbor snapshot has 70 tasks.",
      "metrics": "Per-task executable scientific verifiers.",
      "execution": "Harbor terminal/scientific agent with task-specific resources.",
      "limitations": "Resource-heavy tasks and continuously versioned releases; assess reproducibility and licenses per task.",
      "priority": "P3",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-science",
      "catalogUrl": "https://hub.harborframework.com/datasets/terminal-bench-science/terminal-bench-science",
      "companies": [
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "FutureHouse",
          "url": "https://www.futurehouse.org/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Sakana AI",
          "url": "https://sakana.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "terminal-bench-science/terminal-bench-science",
          "url": "https://hub.harborframework.com/datasets/terminal-bench-science/terminal-bench-science",
          "tasks": 70,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "libero",
      "profileId": "B104",
      "name": "LIBERO",
      "category": "Robotics and CAD",
      "description": "Language-conditioned manipulation and transfer across task suites.",
      "metrics": "Simulator task success and transfer/continual-learning metrics.",
      "execution": "Robot policy adapter to the benchmark embodiment.",
      "limitations": "Direct only for compatible policies; comparing humanoids needs an embodiment adapter and does not establish real-world deployment quality.",
      "priority": "P3",
      "maturity": "Established",
      "sourceUrl": "https://github.com/Lifelong-Robot-Learning/LIBERO",
      "catalogUrl": "https://huggingface.co/datasets/yifengzhu-hf/LIBERO-datasets",
      "companies": [
        {
          "name": "1X",
          "url": "https://www.1x.tech/neo",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Figure",
          "url": "https://www.figure.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Physical Intelligence",
          "url": "https://www.physicalintelligence.company/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Sanctuary AI",
          "url": "https://sanctuary.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Skild AI",
          "url": "https://www.skild.ai/",
          "fit": "Research / simulator adaptation"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "calvin",
      "profileId": "B105",
      "name": "CALVIN",
      "category": "Robotics and CAD",
      "description": "Long-horizon chains of language-conditioned robot manipulation.",
      "metrics": "Sequential-task completion statistics.",
      "execution": "Policy adapter and CALVIN simulator.",
      "limitations": "Hold observation/action spaces, initial states and task-chain protocol fixed.",
      "priority": "P3",
      "maturity": "Established",
      "sourceUrl": "https://github.com/mees/calvin",
      "catalogUrl": "",
      "companies": [
        {
          "name": "1X",
          "url": "https://www.1x.tech/neo",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Figure",
          "url": "https://www.figure.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Physical Intelligence",
          "url": "https://www.physicalintelligence.company/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Sanctuary AI",
          "url": "https://sanctuary.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Skild AI",
          "url": "https://www.skild.ai/",
          "fit": "Research / simulator adaptation"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "rlbench",
      "profileId": "B106",
      "name": "RLBench",
      "category": "Robotics and CAD",
      "description": "Diverse simulated robot manipulation tasks and variations.",
      "metrics": "Task success under the selected task/split protocol.",
      "execution": "CoppeliaSim/PyRep environment and robot policy interface.",
      "limitations": "RLBench is a suite with many configurations; task selection and action mode are part of the benchmark definition.",
      "priority": "P3",
      "maturity": "Established",
      "sourceUrl": "https://github.com/stepjam/RLBench",
      "catalogUrl": "",
      "companies": [
        {
          "name": "1X",
          "url": "https://www.1x.tech/neo",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Figure",
          "url": "https://www.figure.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Physical Intelligence",
          "url": "https://www.physicalintelligence.company/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Sanctuary AI",
          "url": "https://sanctuary.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Skild AI",
          "url": "https://www.skild.ai/",
          "fit": "Research / simulator adaptation"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "maniskill",
      "profileId": "B107",
      "name": "ManiSkill",
      "category": "Robotics and CAD",
      "description": "Robot manipulation in GPU-parallel simulation.",
      "metrics": "Per-environment task success and generalization.",
      "execution": "Supported simulator, robot embodiment and policy.",
      "limitations": "A framework alone is not a frozen evaluation; specify tasks, versions, seeds and training-data access.",
      "priority": "P3",
      "maturity": "Established simulation suite",
      "sourceUrl": "https://github.com/haosulab/ManiSkill",
      "catalogUrl": "",
      "companies": [
        {
          "name": "1X",
          "url": "https://www.1x.tech/neo",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Figure",
          "url": "https://www.figure.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Physical Intelligence",
          "url": "https://www.physicalintelligence.company/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Sanctuary AI",
          "url": "https://sanctuary.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Skild AI",
          "url": "https://www.skild.ai/",
          "fit": "Research / simulator adaptation"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "robocasa-robocasa365",
      "profileId": "B108",
      "name": "RoboCasa / RoboCasa365",
      "category": "Robotics and CAD",
      "description": "Everyday/kitchen robot manipulation in diverse simulated environments.",
      "metrics": "Task success and scene/task generalization.",
      "execution": "Compatible manipulation policy and RoboCasa runtime.",
      "limitations": "Demonstrations are training resources; separate them from frozen evaluation episodes and unseen scenes.",
      "priority": "P3",
      "maturity": "Established family; newer extension",
      "sourceUrl": "https://github.com/robocasa/robocasa",
      "catalogUrl": "",
      "companies": [
        {
          "name": "1X",
          "url": "https://www.1x.tech/neo",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Figure",
          "url": "https://www.figure.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Physical Intelligence",
          "url": "https://www.physicalintelligence.company/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Sanctuary AI",
          "url": "https://sanctuary.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Skild AI",
          "url": "https://www.skild.ai/",
          "fit": "Research / simulator adaptation"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "cad-bench",
      "profileId": "B109",
      "name": "CAD-Bench",
      "category": "Robotics and CAD",
      "description": "100 parametric FreeCAD-generation tasks in the Harbor package.",
      "metrics": "Task-specific CAD artifact checks.",
      "execution": "Agent able to generate/edit supported CAD artifacts.",
      "limitations": "FreeCAD is not the native format of every CAD vendor; geometry/constraint adapters are needed before comparing systems.",
      "priority": "P3",
      "maturity": "Newer specialist",
      "sourceUrl": "https://hub.harborframework.com/datasets/gnucleus-ai/cad-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/gnucleus-ai/cad-bench",
      "companies": [
        {
          "name": "Autodesk",
          "url": "https://www.autodesk.com/solutions/autodesk-ai",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Onshape",
          "url": "https://www.onshape.com/en/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zoo",
          "url": "https://zoo.dev/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "gnucleus-ai/cad-bench",
          "url": "https://hub.harborframework.com/datasets/gnucleus-ai/cad-bench",
          "tasks": 100,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "gpqa-diamond",
      "profileId": "B110",
      "name": "GPQA Diamond",
      "category": "General model capability",
      "description": "Expert-level science multiple-choice questions; Diamond has 198 questions.",
      "metrics": "Answer accuracy.",
      "execution": "Text model with explicitly declared tools.",
      "limitations": "Gated original data; high science reasoning accuracy does not establish agent reliability.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/Idavidrein/gpqa",
      "catalogUrl": "https://hub.harborframework.com/datasets/gpqa-diamond/gpqa-diamond",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "gpqa-diamond/gpqa-diamond",
          "url": "https://hub.harborframework.com/datasets/gpqa-diamond/gpqa-diamond",
          "tasks": 198,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mmlu-pro",
      "profileId": "B111",
      "name": "MMLU-Pro",
      "category": "General model capability",
      "description": "Broad knowledge/reasoning across academic domains.",
      "metrics": "Multiple-choice accuracy by domain.",
      "execution": "Text model with fixed prompt/answer extraction.",
      "limitations": "Supporting model diagnostic; low differentiation as the only benchmark offered to application companies.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/TIGER-Lab/MMLU-Pro",
      "catalogUrl": "https://huggingface.co/datasets/TIGER-Lab/MMLU-Pro",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "TIGER-Lab/MMLU-Pro",
          "url": "https://huggingface.co/datasets/TIGER-Lab/MMLU-Pro",
          "tasks": null,
          "access": "Public",
          "review": "Selected general-model supporting diagnostic."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "humanity-s-last-exam-hle",
      "profileId": "B112",
      "name": "Humanity's Last Exam (HLE)",
      "category": "General model capability",
      "description": "Challenging expert-written multimodal academic questions.",
      "metrics": "Answer accuracy and applicable calibration measures.",
      "execution": "Model with declared text/image/tool access.",
      "limitations": "Gated data with redistribution restrictions; tool-enabled and closed-book settings must remain separate.",
      "priority": "P2",
      "maturity": "Established frontier diagnostic",
      "sourceUrl": "https://huggingface.co/datasets/cais/hle",
      "catalogUrl": "https://huggingface.co/datasets/cais/hle",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "cais/hle",
          "url": "https://huggingface.co/datasets/cais/hle",
          "tasks": null,
          "access": "Gated",
          "review": "Selected frontier model diagnostic; gated, preserve non-redistribution instructions."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "aime-2026-hmmt-2026",
      "profileId": "B113",
      "name": "AIME 2026 / HMMT 2026",
      "category": "General model capability",
      "description": "Recent competition-math reasoning.",
      "metrics": "Final-answer accuracy under fixed sampling and tool policy.",
      "execution": "Mathematical reasoning model.",
      "limitations": "HF cards list CC-BY-NC-SA-4.0; clarify rights before a commercial evaluation offering. Pin year, competition and exact questions.",
      "priority": "P2",
      "maturity": "Established recurring competitions",
      "sourceUrl": "https://huggingface.co/datasets/MathArena/aime_2026",
      "catalogUrl": "https://huggingface.co/datasets/MathArena/hmmt_feb_2026",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "MathArena/hmmt_feb_2026",
          "url": "https://huggingface.co/datasets/MathArena/hmmt_feb_2026",
          "tasks": null,
          "access": "Public",
          "review": "Selected recent-math companion; noncommercial license tag."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "ifeval",
      "profileId": "B114",
      "name": "IFEval",
      "category": "General model capability",
      "description": "Verifiable instruction-following constraints.",
      "metrics": "Strict/loose prompt-level and instruction-level accuracy.",
      "execution": "Text-generation model or prompt pipeline.",
      "limitations": "Constraint compliance is different from factual or task correctness; keep instruction templates fixed.",
      "priority": "P1",
      "maturity": "Established diagnostic",
      "sourceUrl": "https://github.com/google-research/google-research/tree/master/instruction_following_eval",
      "catalogUrl": "",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Glean",
          "url": "https://www.glean.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Writer",
          "url": "https://writer.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "ifstruct",
      "profileId": "B115",
      "name": "IFStruct",
      "category": "General model capability",
      "description": "JSON/YAML structural compliance across paraphrased schema requests.",
      "metrics": "Structure/schema validity without constrained decoding.",
      "execution": "Model returning structured outputs.",
      "limitations": "The published protocol excludes constrained decoding and does not assess content correctness; label any changed decoding setting.",
      "priority": "P2",
      "maturity": "Newer specialist",
      "sourceUrl": "https://huggingface.co/datasets/LiquidAI/ifstruct-v1.0",
      "catalogUrl": "https://huggingface.co/datasets/LiquidAI/ifstruct-v1.0",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "Composio",
          "url": "https://composio.dev/",
          "fit": "Composite system"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Zapier",
          "url": "https://zapier.com/agents",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "n8n",
          "url": "https://n8n.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "LiquidAI/ifstruct-v1.0",
          "url": "https://huggingface.co/datasets/LiquidAI/ifstruct-v1.0",
          "tasks": null,
          "access": "Public",
          "review": "Selected structured-output diagnostic; no constrained decoding in original protocol."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "arc-agi-2",
      "profileId": "B116",
      "name": "ARC-AGI-2",
      "category": "General model capability",
      "description": "Novel abstract visual-grid transformations.",
      "metrics": "Exact task accuracy at a reported inference-cost budget.",
      "execution": "Reasoning system using the official task protocol.",
      "limitations": "Public, semi-private and private evaluation sets differ. The Harbor 167-task package is not the full official private evaluation.",
      "priority": "P2",
      "maturity": "Established frontier diagnostic",
      "sourceUrl": "https://arcprize.org/arc-agi/2",
      "catalogUrl": "https://hub.harborframework.com/datasets/arcprize/arc-agi-2",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Harbor",
          "name": "arcprize/arc-agi-2",
          "url": "https://hub.harborframework.com/datasets/arcprize/arc-agi-2",
          "tasks": 167,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "gsm8k",
      "profileId": "B117",
      "name": "GSM8K",
      "category": "General model capability",
      "description": "Grade-school multi-step arithmetic word problems.",
      "metrics": "Exact-answer accuracy.",
      "execution": "Text model with specified calculator/code access.",
      "limitations": "Famous but widely exposed and often saturated; keep for regression, not the central value proposition of a new evaluation service.",
      "priority": "P3",
      "maturity": "Historical baseline",
      "sourceUrl": "https://huggingface.co/datasets/openai/gsm8k",
      "catalogUrl": "https://huggingface.co/datasets/openai/gsm8k",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "openai/gsm8k",
          "url": "https://huggingface.co/datasets/openai/gsm8k",
          "tasks": null,
          "access": "Public",
          "review": "Historical regression baseline; not a differentiated agent benchmark."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "docvqa",
      "profileId": "B118",
      "name": "DocVQA",
      "category": "Document parsing and extraction",
      "description": "Question answering over document images.",
      "metrics": "ANLS or challenge-specific answer metrics.",
      "execution": "Document-image question-answering endpoint.",
      "limitations": "Specify single-document, collection, infographics or the new challenge track. This tests answering questions, not complete PDF parsing.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://www.docvqa.org/",
      "catalogUrl": "",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "chartqa",
      "profileId": "B119",
      "name": "ChartQA",
      "category": "Document parsing and extraction",
      "description": "Visual and numerical reasoning about chart images.",
      "metrics": "Relaxed answer accuracy under the published protocol.",
      "execution": "Vision-language model or chart-aware document assistant.",
      "limitations": "Report human-written and generated question subsets; preserve chart resolution and access to underlying tables.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://github.com/vis-nlp/ChartQA",
      "catalogUrl": "",
      "companies": [
        {
          "name": "ABBYY",
          "url": "https://www.abbyy.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Extend",
          "url": "https://www.extend.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Instabase",
          "url": "https://instabase.com/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "LlamaIndex",
          "url": "https://www.llamaindex.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Reducto",
          "url": "https://reducto.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rossum",
          "url": "https://rossum.ai/",
          "fit": "Direct document API after output mapping"
        },
        {
          "name": "Unstructured",
          "url": "https://unstructured.io/",
          "fit": "Direct document API after output mapping"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mathvista",
      "profileId": "B120",
      "name": "MathVista",
      "category": "Multimodal understanding",
      "description": "Mathematical reasoning in diagrams, figures and other visual contexts.",
      "metrics": "Answer accuracy by task/category.",
      "execution": "Image-and-text reasoning model.",
      "limitations": "Not a substitute for spreadsheet execution or financial-document retrieval; label tool and visual input access.",
      "priority": "P2",
      "maturity": "Established",
      "sourceUrl": "https://huggingface.co/datasets/AI4Math/MathVista",
      "catalogUrl": "https://huggingface.co/datasets/AI4Math/MathVista",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Hebbia",
          "url": "https://www.hebbia.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Rogo",
          "url": "https://www.rogo.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "math-500",
      "profileId": "B121",
      "name": "MATH-500",
      "category": "General model capability",
      "description": "A 500-problem subset of competition mathematics.",
      "metrics": "Normalized final-answer accuracy.",
      "execution": "Mathematical reasoning model.",
      "limitations": "Widely exposed public test data; recent competition sets can provide a fresher complementary signal.",
      "priority": "P3",
      "maturity": "Historical comparison baseline",
      "sourceUrl": "https://huggingface.co/datasets/HuggingFaceH4/MATH-500",
      "catalogUrl": "https://huggingface.co/datasets/HuggingFaceH4/MATH-500",
      "companies": [
        {
          "name": "Alibaba Qwen",
          "url": "https://qwen.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Anthropic",
          "url": "https://www.anthropic.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Cohere",
          "url": "https://cohere.com/",
          "fit": "Direct component API + selected system adapters"
        },
        {
          "name": "DeepSeek",
          "url": "https://www.deepseek.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Google DeepMind",
          "url": "https://deepmind.google/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Meta",
          "url": "https://ai.meta.com/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "MiniMax",
          "url": "https://www.minimax.io/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Mistral",
          "url": "https://mistral.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "OpenAI",
          "url": "https://openai.com/index/browsecomp/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "Z.ai",
          "url": "https://z.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        },
        {
          "name": "xAI",
          "url": "https://x.ai/",
          "fit": "Capability-aligned; adapter/access to qualify"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "behavior-behavior-1k",
      "profileId": "B122",
      "name": "BEHAVIOR / BEHAVIOR-1K",
      "category": "Robotics and CAD",
      "description": "Long-horizon everyday household activities and object interactions.",
      "metrics": "Task completion in the specified simulator/task suite.",
      "execution": "Compatible robot policy and supported simulation assets.",
      "limitations": "Licensing, compute and embodiment requirements differ from text-agent evaluations; demonstration datasets and evaluation episodes must remain separate.",
      "priority": "P3",
      "maturity": "Established simulation family",
      "sourceUrl": "https://behavior.stanford.edu/",
      "catalogUrl": "",
      "companies": [
        {
          "name": "1X",
          "url": "https://www.1x.tech/neo",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Figure",
          "url": "https://www.figure.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "NVIDIA",
          "url": "https://www.nvidia.com/en-us/industries/robotics/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Physical Intelligence",
          "url": "https://www.physicalintelligence.company/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Sanctuary AI",
          "url": "https://sanctuary.ai/",
          "fit": "Research / simulator adaptation"
        },
        {
          "name": "Skild AI",
          "url": "https://www.skild.ai/",
          "fit": "Research / simulator adaptation"
        }
      ],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "humaneval",
      "profileId": null,
      "name": "HumanEval",
      "category": "Coding",
      "description": "Python function generation with original executable tests.",
      "metrics": "Native task success, errors, latency and recorded cost.",
      "execution": "Blobfish-hosted Harbor harness on GKE; frozen task subset.",
      "limitations": "The published pilot is a subset. Inspect the run protocol before comparing scores.",
      "priority": "P1",
      "maturity": "Hosted task package",
      "sourceUrl": "https://github.com/openai/human-eval",
      "catalogUrl": "",
      "companies": [],
      "sources": [],
      "hostedId": "humaneval",
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "mbpp",
      "profileId": null,
      "name": "MBPP",
      "category": "Coding",
      "description": "Python programming tasks evaluated against source-defined tests.",
      "metrics": "Native task success, errors, latency and recorded cost.",
      "execution": "Blobfish-hosted Harbor harness on GKE; frozen task subset.",
      "limitations": "The published pilot is a subset. Inspect the run protocol before comparing scores.",
      "priority": "P1",
      "maturity": "Hosted task package",
      "sourceUrl": "https://github.com/google-research/google-research/tree/master/mbpp",
      "catalogUrl": "",
      "companies": [],
      "sources": [],
      "hostedId": "mbpp",
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "devopsbench",
      "profileId": null,
      "name": "DevOpsBench-100",
      "category": "Software operations",
      "description": "Long-horizon DevOps/SRE work over one executable NovaCart world: root-cause analysis, code changes, canaried deploys, migrations, flags, and incident closure.",
      "metrics": "Native task success, errors, latency and recorded cost.",
      "execution": "Blobfish-hosted Harbor harness on GKE; frozen task subset.",
      "limitations": "The published pilot is a subset. Inspect the run protocol before comparing scores.",
      "priority": "P1",
      "maturity": "Hosted task package",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/devopsbench-100",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/devopsbench-100",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/devopsbench-100",
          "url": "https://hub.harborframework.com/datasets/blobfishai/devopsbench-100",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": "devopsbench",
      "productHref": "/benchmarks/devopsbench-100",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "mind2web",
      "profileId": null,
      "name": "Mind2Web (offline)",
      "category": "Browser automation",
      "description": "Offline web action prediction on recorded website interactions.",
      "metrics": "Element selection and action prediction under the original offline protocol.",
      "execution": "Recorded web interactions; not a live-web success evaluation.",
      "limitations": "Distinct from Online-Mind2Web and Mind2Web 2. Scores cannot be exchanged across these datasets.",
      "priority": "P2",
      "maturity": "Established offline benchmark",
      "sourceUrl": "https://github.com/OSU-NLP-Group/Mind2Web",
      "catalogUrl": "https://huggingface.co/datasets/osunlp/Mind2Web",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-terminal-bench-terminal-bench-2",
      "profileId": null,
      "name": "terminal-bench/terminal-bench-2",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2",
      "catalogUrl": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "terminal-bench/terminal-bench-2",
          "url": "https://hub.harborframework.com/datasets/terminal-bench/terminal-bench-2",
          "tasks": 89,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-scale-ai-swe-bench-pro",
      "profileId": null,
      "name": "scale-ai/swe-bench-pro",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/scale-ai/swe-bench-pro",
      "catalogUrl": "https://hub.harborframework.com/datasets/scale-ai/swe-bench-pro",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "scale-ai/swe-bench-pro",
          "url": "https://hub.harborframework.com/datasets/scale-ai/swe-bench-pro",
          "tasks": 731,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-harbor-hello-world",
      "profileId": null,
      "name": "harbor/hello-world",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/harbor/hello-world",
      "catalogUrl": "https://hub.harborframework.com/datasets/harbor/hello-world",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "harbor/hello-world",
          "url": "https://hub.harborframework.com/datasets/harbor/hello-world",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-openthoughts-tblite",
      "profileId": null,
      "name": "openthoughts/openthoughts-tblite",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/openthoughts-tblite",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/openthoughts-tblite",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/openthoughts-tblite",
          "url": "https://hub.harborframework.com/datasets/openthoughts/openthoughts-tblite",
          "tasks": 100,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-datacurve-deep-swe-1-1",
      "profileId": null,
      "name": "datacurve/deep-swe-1-1",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/datacurve/deep-swe-1-1",
      "catalogUrl": "https://hub.harborframework.com/datasets/datacurve/deep-swe-1-1",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "datacurve/deep-swe-1-1",
          "url": "https://hub.harborframework.com/datasets/datacurve/deep-swe-1-1",
          "tasks": 113,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-datacurve-deep-swe",
      "profileId": null,
      "name": "datacurve/deep-swe",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/datacurve/deep-swe",
      "catalogUrl": "https://hub.harborframework.com/datasets/datacurve/deep-swe",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "datacurve/deep-swe",
          "url": "https://hub.harborframework.com/datasets/datacurve/deep-swe",
          "tasks": 113,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-snorkel-ai-senior-swe-bench-v2026-06",
      "profileId": null,
      "name": "snorkel-ai/senior-swe-bench-v2026.06",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/snorkel-ai/senior-swe-bench-v2026.06",
      "catalogUrl": "https://hub.harborframework.com/datasets/snorkel-ai/senior-swe-bench-v2026.06",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "snorkel-ai/senior-swe-bench-v2026.06",
          "url": "https://hub.harborframework.com/datasets/snorkel-ai/senior-swe-bench-v2026.06",
          "tasks": 50,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-abundant-swe-marathon",
      "profileId": null,
      "name": "abundant/swe-marathon",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/abundant/swe-marathon",
      "catalogUrl": "https://hub.harborframework.com/datasets/abundant/swe-marathon",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "abundant/swe-marathon",
          "url": "https://hub.harborframework.com/datasets/abundant/swe-marathon",
          "tasks": 20,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-enterprise-bench-l1-l2-bench",
      "profileId": null,
      "name": "Enterprise-Bench/l1-l2-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/Enterprise-Bench/l1-l2-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/Enterprise-Bench/l1-l2-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "Enterprise-Bench/l1-l2-bench",
          "url": "https://hub.harborframework.com/datasets/Enterprise-Bench/l1-l2-bench",
          "tasks": 14,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-harbor-index-harbor-index",
      "profileId": null,
      "name": "harbor-index/harbor-index",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/harbor-index/harbor-index",
      "catalogUrl": "https://hub.harborframework.com/datasets/harbor-index/harbor-index",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "harbor-index/harbor-index",
          "url": "https://hub.harborframework.com/datasets/harbor-index/harbor-index",
          "tasks": 80,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-maxbittker-runebench",
      "profileId": null,
      "name": "maxbittker/runebench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/maxbittker/runebench",
      "catalogUrl": "https://hub.harborframework.com/datasets/maxbittker/runebench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "maxbittker/runebench",
          "url": "https://hub.harborframework.com/datasets/maxbittker/runebench",
          "tasks": 32,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-gabeorlanski-slopcodebench",
      "profileId": null,
      "name": "gabeorlanski/slopcodebench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/gabeorlanski/slopcodebench",
      "catalogUrl": "https://hub.harborframework.com/datasets/gabeorlanski/slopcodebench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "gabeorlanski/slopcodebench",
          "url": "https://hub.harborframework.com/datasets/gabeorlanski/slopcodebench",
          "tasks": 36,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ibragim-badertdinov-swe-rebench-07-2026",
      "profileId": null,
      "name": "ibragim-badertdinov/swe-rebench-07-2026",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ibragim-badertdinov/swe-rebench-07-2026",
      "catalogUrl": "https://hub.harborframework.com/datasets/ibragim-badertdinov/swe-rebench-07-2026",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ibragim-badertdinov/swe-rebench-07-2026",
          "url": "https://hub.harborframework.com/datasets/ibragim-badertdinov/swe-rebench-07-2026",
          "tasks": 111,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-terminal-bench-pro-terminal-bench-pro",
      "profileId": null,
      "name": "terminal-bench-pro/terminal-bench-pro",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/terminal-bench-pro/terminal-bench-pro",
      "catalogUrl": "https://hub.harborframework.com/datasets/terminal-bench-pro/terminal-bench-pro",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "terminal-bench-pro/terminal-bench-pro",
          "url": "https://hub.harborframework.com/datasets/terminal-bench-pro/terminal-bench-pro",
          "tasks": 200,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-orca-bench-orca-bench",
      "profileId": null,
      "name": "orca-bench/orca-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/orca-bench/orca-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/orca-bench/orca-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "orca-bench/orca-bench",
          "url": "https://hub.harborframework.com/datasets/orca-bench/orca-bench",
          "tasks": 755,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tmax-tmax-15k-harbor",
      "profileId": null,
      "name": "tmax/TMax-15K-Harbor",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/tmax/TMax-15K-Harbor",
      "catalogUrl": "https://hub.harborframework.com/datasets/tmax/TMax-15K-Harbor",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tmax/TMax-15K-Harbor",
          "url": "https://hub.harborframework.com/datasets/tmax/TMax-15K-Harbor",
          "tasks": 14601,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-aime-aime",
      "profileId": null,
      "name": "aime/aime",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/aime/aime",
      "catalogUrl": "https://hub.harborframework.com/datasets/aime/aime",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "aime/aime",
          "url": "https://hub.harborframework.com/datasets/aime/aime",
          "tasks": 60,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-vmax-modal-modal-port-v1-eval-patched",
      "profileId": null,
      "name": "vmax-modal/modal-port-v1-eval-patched",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/vmax-modal/modal-port-v1-eval-patched",
      "catalogUrl": "https://hub.harborframework.com/datasets/vmax-modal/modal-port-v1-eval-patched",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "vmax-modal/modal-port-v1-eval-patched",
          "url": "https://hub.harborframework.com/datasets/vmax-modal/modal-port-v1-eval-patched",
          "tasks": 427,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-mercor-apex-agents-1-1",
      "profileId": null,
      "name": "mercor/apex-agents-1-1",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/mercor/apex-agents-1-1",
      "catalogUrl": "https://hub.harborframework.com/datasets/mercor/apex-agents-1-1",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "mercor/apex-agents-1-1",
          "url": "https://hub.harborframework.com/datasets/mercor/apex-agents-1-1",
          "tasks": 240,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-swe-rebench-swe-rebench-leaderboard",
      "profileId": null,
      "name": "swe-rebench/swe-rebench-leaderboard",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/swe-rebench/swe-rebench-leaderboard",
      "catalogUrl": "https://hub.harborframework.com/datasets/swe-rebench/swe-rebench-leaderboard",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "swe-rebench/swe-rebench-leaderboard",
          "url": "https://hub.harborframework.com/datasets/swe-rebench/swe-rebench-leaderboard",
          "tasks": 860,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tencent-autocodebench",
      "profileId": null,
      "name": "tencent/autocodebench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tencent/autocodebench",
      "catalogUrl": "https://hub.harborframework.com/datasets/tencent/autocodebench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tencent/autocodebench",
          "url": "https://hub.harborframework.com/datasets/tencent/autocodebench",
          "tasks": 200,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-scale-ai-swe-atlas-tw",
      "profileId": null,
      "name": "scale-ai/swe-atlas-tw",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-tw",
      "catalogUrl": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-tw",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "scale-ai/swe-atlas-tw",
          "url": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-tw",
          "tasks": 90,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-bencalvert04-programbench",
      "profileId": null,
      "name": "bencalvert04/programbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/bencalvert04/programbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/bencalvert04/programbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "bencalvert04/programbench",
          "url": "https://hub.harborframework.com/datasets/bencalvert04/programbench",
          "tasks": 200,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-hack-verifiable-environments-hv-terminal-bench-2-1",
      "profileId": null,
      "name": "hack-verifiable-environments/hv-terminal-bench-2-1",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/hack-verifiable-environments/hv-terminal-bench-2-1",
      "catalogUrl": "https://hub.harborframework.com/datasets/hack-verifiable-environments/hv-terminal-bench-2-1",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "hack-verifiable-environments/hv-terminal-bench-2-1",
          "url": "https://hub.harborframework.com/datasets/hack-verifiable-environments/hv-terminal-bench-2-1",
          "tasks": 89,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-sldbench-sldbench",
      "profileId": null,
      "name": "sldbench/sldbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/sldbench/sldbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/sldbench/sldbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "sldbench/sldbench",
          "url": "https://hub.harborframework.com/datasets/sldbench/sldbench",
          "tasks": 8,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-livecodebench-livecodebench",
      "profileId": null,
      "name": "livecodebench/livecodebench",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/livecodebench/livecodebench",
      "catalogUrl": "https://hub.harborframework.com/datasets/livecodebench/livecodebench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "livecodebench/livecodebench",
          "url": "https://hub.harborframework.com/datasets/livecodebench/livecodebench",
          "tasks": 100,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-red-hat-ai-haiku-hard",
      "profileId": null,
      "name": "red-hat-ai/haiku-hard",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/red-hat-ai/haiku-hard",
      "catalogUrl": "https://hub.harborframework.com/datasets/red-hat-ai/haiku-hard",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "red-hat-ai/haiku-hard",
          "url": "https://hub.harborframework.com/datasets/red-hat-ai/haiku-hard",
          "tasks": 138,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-frontis-agent-obs-core",
      "profileId": null,
      "name": "frontis/agent-obs-core",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/frontis/agent-obs-core",
      "catalogUrl": "https://hub.harborframework.com/datasets/frontis/agent-obs-core",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "frontis/agent-obs-core",
          "url": "https://hub.harborframework.com/datasets/frontis/agent-obs-core",
          "tasks": 3,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-actava-ai-chi-bench",
      "profileId": null,
      "name": "actava-ai/chi-bench",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/actava-ai/chi-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/actava-ai/chi-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "actava-ai/chi-bench",
          "url": "https://hub.harborframework.com/datasets/actava-ai/chi-bench",
          "tasks": 78,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-nl2repobench-nl2repobench",
      "profileId": null,
      "name": "nl2repobench/nl2repobench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/nl2repobench/nl2repobench",
      "catalogUrl": "https://hub.harborframework.com/datasets/nl2repobench/nl2repobench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "nl2repobench/nl2repobench",
          "url": "https://hub.harborframework.com/datasets/nl2repobench/nl2repobench",
          "tasks": 104,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-quesma-otel-bench",
      "profileId": null,
      "name": "quesma/otel-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/quesma/otel-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/quesma/otel-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "quesma/otel-bench",
          "url": "https://hub.harborframework.com/datasets/quesma/otel-bench",
          "tasks": 26,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-replicationbench-replicationbench",
      "profileId": null,
      "name": "replicationbench/replicationbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/replicationbench/replicationbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/replicationbench/replicationbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "replicationbench/replicationbench",
          "url": "https://hub.harborframework.com/datasets/replicationbench/replicationbench",
          "tasks": 90,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-scienceagentbench-scienceagentbench",
      "profileId": null,
      "name": "scienceagentbench/scienceagentbench",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/scienceagentbench/scienceagentbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/scienceagentbench/scienceagentbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "scienceagentbench/scienceagentbench",
          "url": "https://hub.harborframework.com/datasets/scienceagentbench/scienceagentbench",
          "tasks": 102,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-userbench-userbench",
      "profileId": null,
      "name": "userbench/UserBench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/userbench/UserBench",
      "catalogUrl": "https://hub.harborframework.com/datasets/userbench/UserBench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "userbench/UserBench",
          "url": "https://hub.harborframework.com/datasets/userbench/UserBench",
          "tasks": 620,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-aarr-aarri-bench",
      "profileId": null,
      "name": "aarr/aarri-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/aarr/aarri-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/aarr/aarri-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "aarr/aarri-bench",
          "url": "https://hub.harborframework.com/datasets/aarr/aarri-bench",
          "tasks": 82,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-abundant-swe-gen-java",
      "profileId": null,
      "name": "abundant/swe-gen-java",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-java",
      "catalogUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-java",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "abundant/swe-gen-java",
          "url": "https://hub.harborframework.com/datasets/abundant/swe-gen-java",
          "tasks": 1000,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-abundant-swe-gen-rust",
      "profileId": null,
      "name": "abundant/swe-gen-rust",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-rust",
      "catalogUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-rust",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "abundant/swe-gen-rust",
          "url": "https://hub.harborframework.com/datasets/abundant/swe-gen-rust",
          "tasks": 1000,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-bauerjustin-terminal-bench-3-test",
      "profileId": null,
      "name": "bauerjustin/terminal-bench-3-test",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/bauerjustin/terminal-bench-3-test",
      "catalogUrl": "https://hub.harborframework.com/datasets/bauerjustin/terminal-bench-3-test",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "bauerjustin/terminal-bench-3-test",
          "url": "https://hub.harborframework.com/datasets/bauerjustin/terminal-bench-3-test",
          "tasks": 64,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-novitaai-tb21-file-recovery",
      "profileId": null,
      "name": "NovitaAI/tb21-file-recovery",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-file-recovery",
      "catalogUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-file-recovery",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "NovitaAI/tb21-file-recovery",
          "url": "https://hub.harborframework.com/datasets/NovitaAI/tb21-file-recovery",
          "tasks": 9,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-quesma-compilebench",
      "profileId": null,
      "name": "quesma/compilebench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/quesma/compilebench",
      "catalogUrl": "https://hub.harborframework.com/datasets/quesma/compilebench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "quesma/compilebench",
          "url": "https://hub.harborframework.com/datasets/quesma/compilebench",
          "tasks": 15,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-rounakbende10-rh-swe-bench",
      "profileId": null,
      "name": "rounakbende10/rh-swe-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/rounakbende10/rh-swe-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/rounakbende10/rh-swe-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "rounakbende10/rh-swe-bench",
          "url": "https://hub.harborframework.com/datasets/rounakbende10/rh-swe-bench",
          "tasks": 341,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-vals-financeagent",
      "profileId": null,
      "name": "vals/financeagent",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/vals/financeagent",
      "catalogUrl": "https://hub.harborframework.com/datasets/vals/financeagent",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "vals/financeagent",
          "url": "https://hub.harborframework.com/datasets/vals/financeagent",
          "tasks": 50,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-agentscope-ai-pawbench",
      "profileId": null,
      "name": "agentscope-ai/pawbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/agentscope-ai/pawbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/agentscope-ai/pawbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "agentscope-ai/pawbench",
          "url": "https://hub.harborframework.com/datasets/agentscope-ai/pawbench",
          "tasks": 150,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-novitaai-tb21-data-science",
      "profileId": null,
      "name": "NovitaAI/tb21-data-science",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-data-science",
      "catalogUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-data-science",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "NovitaAI/tb21-data-science",
          "url": "https://hub.harborframework.com/datasets/NovitaAI/tb21-data-science",
          "tasks": 26,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-reasoning-gym-reasoning-gym-easy",
      "profileId": null,
      "name": "reasoning-gym/reasoning-gym-easy",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/reasoning-gym/reasoning-gym-easy",
      "catalogUrl": "https://hub.harborframework.com/datasets/reasoning-gym/reasoning-gym-easy",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "reasoning-gym/reasoning-gym-easy",
          "url": "https://hub.harborframework.com/datasets/reasoning-gym/reasoning-gym-easy",
          "tasks": 288,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-scale-ai-swe-atlas-rf",
      "profileId": null,
      "name": "scale-ai/swe-atlas-rf",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-rf",
      "catalogUrl": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-rf",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "scale-ai/swe-atlas-rf",
          "url": "https://hub.harborframework.com/datasets/scale-ai/swe-atlas-rf",
          "tasks": 70,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-swe-bench-swe-smith",
      "profileId": null,
      "name": "swe-bench/swe-smith",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/swe-bench/swe-smith",
      "catalogUrl": "https://hub.harborframework.com/datasets/swe-bench/swe-smith",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "swe-bench/swe-smith",
          "url": "https://hub.harborframework.com/datasets/swe-bench/swe-smith",
          "tasks": 100,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-abundant-swe-gen-cpp",
      "profileId": null,
      "name": "abundant/swe-gen-cpp",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-cpp",
      "catalogUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-cpp",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "abundant/swe-gen-cpp",
          "url": "https://hub.harborframework.com/datasets/abundant/swe-gen-cpp",
          "tasks": 999,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-factory-ai-legacy-bench",
      "profileId": null,
      "name": "factory-ai/legacy-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/factory-ai/legacy-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/factory-ai/legacy-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "factory-ai/legacy-bench",
          "url": "https://hub.harborframework.com/datasets/factory-ai/legacy-bench",
          "tasks": 10,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-cais-swebenchpro",
      "profileId": null,
      "name": "cais/swebenchpro",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/cais/swebenchpro",
      "catalogUrl": "https://hub.harborframework.com/datasets/cais/swebenchpro",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "cais/swebenchpro",
          "url": "https://hub.harborframework.com/datasets/cais/swebenchpro",
          "tasks": 731,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-factorybench-100",
      "profileId": null,
      "name": "FactoryBench-100",
      "category": "Manufacturing operations",
      "description": "Realistic manufacturing decisions framed as high-level employee requests. Agents must discover the relevant Oracle Fusion, Gmail, Drive, Sheets, and Slack evidence, reconcile records, calculate and compare options, then execute only the supported outcome.",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/factorybench-100",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/factorybench-100",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/factorybench-100",
          "url": "https://hub.harborframework.com/datasets/blobfishai/factorybench-100",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": "/benchmarks/factorybench-100",
      "productResultState": "published"
    },
    {
      "id": "harbor-cohere-terminal-bench-aai",
      "profileId": null,
      "name": "cohere/terminal_bench_aai",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/cohere/terminal_bench_aai",
      "catalogUrl": "https://hub.harborframework.com/datasets/cohere/terminal_bench_aai",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "cohere/terminal_bench_aai",
          "url": "https://hub.harborframework.com/datasets/cohere/terminal_bench_aai",
          "tasks": 44,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-frontis-generated-basic-production-traces",
      "profileId": null,
      "name": "frontis/generated-basic-production-traces",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/frontis/generated-basic-production-traces",
      "catalogUrl": "https://hub.harborframework.com/datasets/frontis/generated-basic-production-traces",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "frontis/generated-basic-production-traces",
          "url": "https://hub.harborframework.com/datasets/frontis/generated-basic-production-traces",
          "tasks": 57,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-kgmon-deepsearchqa",
      "profileId": null,
      "name": "kgmon/deepsearchqa",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/kgmon/deepsearchqa",
      "catalogUrl": "https://hub.harborframework.com/datasets/kgmon/deepsearchqa",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "kgmon/deepsearchqa",
          "url": "https://hub.harborframework.com/datasets/kgmon/deepsearchqa",
          "tasks": 900,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tinycomputerai-bun-server-bench",
      "profileId": null,
      "name": "tinycomputerai/bun-server-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/tinycomputerai/bun-server-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/tinycomputerai/bun-server-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tinycomputerai/bun-server-bench",
          "url": "https://hub.harborframework.com/datasets/tinycomputerai/bun-server-bench",
          "tasks": 50,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-agentic-labs-erp-bench",
      "profileId": null,
      "name": "agentic-labs/erp-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/agentic-labs/erp-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/agentic-labs/erp-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "agentic-labs/erp-bench",
          "url": "https://hub.harborframework.com/datasets/agentic-labs/erp-bench",
          "tasks": 300,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-mmtb-multimedia-terminalbench",
      "profileId": null,
      "name": "mmtb/multimedia-terminalbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/mmtb/multimedia-terminalbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/mmtb/multimedia-terminalbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "mmtb/multimedia-terminalbench",
          "url": "https://hub.harborframework.com/datasets/mmtb/multimedia-terminalbench",
          "tasks": 105,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-scale-ai-hil-bench",
      "profileId": null,
      "name": "scale-ai/hil-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/scale-ai/hil-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/scale-ai/hil-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "scale-ai/hil-bench",
          "url": "https://hub.harborframework.com/datasets/scale-ai/hil-bench",
          "tasks": 600,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-abundant-swe-gen-go",
      "profileId": null,
      "name": "abundant/swe-gen-go",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-go",
      "catalogUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-go",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "abundant/swe-gen-go",
          "url": "https://hub.harborframework.com/datasets/abundant/swe-gen-go",
          "tasks": 1000,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-algotune-algotune",
      "profileId": null,
      "name": "algotune/algotune",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/algotune/algotune",
      "catalogUrl": "https://hub.harborframework.com/datasets/algotune/algotune",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "algotune/algotune",
          "url": "https://hub.harborframework.com/datasets/algotune/algotune",
          "tasks": 154,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-meta-mlgym-bench",
      "profileId": null,
      "name": "meta/mlgym-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/meta/mlgym-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/meta/mlgym-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "meta/mlgym-bench",
          "url": "https://hub.harborframework.com/datasets/meta/mlgym-bench",
          "tasks": 12,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-orinlabs-horizon-public",
      "profileId": null,
      "name": "orinlabs/horizon-public",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/orinlabs/horizon-public",
      "catalogUrl": "https://hub.harborframework.com/datasets/orinlabs/horizon-public",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "orinlabs/horizon-public",
          "url": "https://hub.harborframework.com/datasets/orinlabs/horizon-public",
          "tasks": 3,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-userbench-userbench-train400",
      "profileId": null,
      "name": "userbench/UserBench-train400",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/userbench/UserBench-train400",
      "catalogUrl": "https://hub.harborframework.com/datasets/userbench/UserBench-train400",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "userbench/UserBench-train400",
          "url": "https://hub.harborframework.com/datasets/userbench/UserBench-train400",
          "tasks": 620,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-bigcode-humanevalfix",
      "profileId": null,
      "name": "bigcode/humanevalfix",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/bigcode/humanevalfix",
      "catalogUrl": "https://hub.harborframework.com/datasets/bigcode/humanevalfix",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "bigcode/humanevalfix",
          "url": "https://hub.harborframework.com/datasets/bigcode/humanevalfix",
          "tasks": 164,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-counselbench-100",
      "profileId": null,
      "name": "CounselBench-100",
      "category": "Legal operations",
      "description": "Long-horizon legal work across evidence rooms, custody checks, source-grounded findings, and final deliverables.",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/counselbench-100",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/counselbench-100",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/counselbench-100",
          "url": "https://hub.harborframework.com/datasets/blobfishai/counselbench-100",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": "/benchmarks/counselbench-100",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "harbor-futurehouse-bixbench-cli",
      "profileId": null,
      "name": "futurehouse/bixbench-cli",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/futurehouse/bixbench-cli",
      "catalogUrl": "https://hub.harborframework.com/datasets/futurehouse/bixbench-cli",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "futurehouse/bixbench-cli",
          "url": "https://hub.harborframework.com/datasets/futurehouse/bixbench-cli",
          "tasks": 205,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-quixbugs-quixbugs",
      "profileId": null,
      "name": "quixbugs/quixbugs",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/quixbugs/quixbugs",
      "catalogUrl": "https://hub.harborframework.com/datasets/quixbugs/quixbugs",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "quixbugs/quixbugs",
          "url": "https://hub.harborframework.com/datasets/quixbugs/quixbugs",
          "tasks": 80,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-reasoning-gym-reasoning-gym-hard",
      "profileId": null,
      "name": "reasoning-gym/reasoning-gym-hard",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/reasoning-gym/reasoning-gym-hard",
      "catalogUrl": "https://hub.harborframework.com/datasets/reasoning-gym/reasoning-gym-hard",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "reasoning-gym/reasoning-gym-hard",
          "url": "https://hub.harborframework.com/datasets/reasoning-gym/reasoning-gym-hard",
          "tasks": 288,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-webgen-bench-webgen-bench",
      "profileId": null,
      "name": "webgen-bench/webgen-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/webgen-bench/webgen-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/webgen-bench/webgen-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "webgen-bench/webgen-bench",
          "url": "https://hub.harborframework.com/datasets/webgen-bench/webgen-bench",
          "tasks": 101,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-yanagiorigami-frontier-cs",
      "profileId": null,
      "name": "yanagiorigami/frontier-cs",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/yanagiorigami/frontier-cs",
      "catalogUrl": "https://hub.harborframework.com/datasets/yanagiorigami/frontier-cs",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "yanagiorigami/frontier-cs",
          "url": "https://hub.harborframework.com/datasets/yanagiorigami/frontier-cs",
          "tasks": 172,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-bauerjustin-tb3-preview-v2",
      "profileId": null,
      "name": "bauerjustin/tb3-preview-v2",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/bauerjustin/tb3-preview-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/bauerjustin/tb3-preview-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "bauerjustin/tb3-preview-v2",
          "url": "https://hub.harborframework.com/datasets/bauerjustin/tb3-preview-v2",
          "tasks": 111,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-bigcode-bigcodebench-hard-complete",
      "profileId": null,
      "name": "bigcode/bigcodebench-hard-complete",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/bigcode/bigcodebench-hard-complete",
      "catalogUrl": "https://hub.harborframework.com/datasets/bigcode/bigcodebench-hard-complete",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "bigcode/bigcodebench-hard-complete",
          "url": "https://hub.harborframework.com/datasets/bigcode/bigcodebench-hard-complete",
          "tasks": 145,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-binary-audit-binary-audit",
      "profileId": null,
      "name": "binary-audit/binary-audit",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/binary-audit/binary-audit",
      "catalogUrl": "https://hub.harborframework.com/datasets/binary-audit/binary-audit",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "binary-audit/binary-audit",
          "url": "https://hub.harborframework.com/datasets/binary-audit/binary-audit",
          "tasks": 46,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-salesbench-100",
      "profileId": null,
      "name": "SalesBench-100",
      "category": "Sales operations",
      "description": "Long-horizon revenue work across Salesforce, HubSpot, Gong, seeded account rooms, controlled CRM mutations, and executive deliverables.",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/salesbench-100",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/salesbench-100",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/salesbench-100",
          "url": "https://hub.harborframework.com/datasets/blobfishai/salesbench-100",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": "/benchmarks/salesbench-100",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "harbor-islo-labs-reward-hack-bench",
      "profileId": null,
      "name": "islo-labs/reward-hack-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/islo-labs/reward-hack-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/islo-labs/reward-hack-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "islo-labs/reward-hack-bench",
          "url": "https://hub.harborframework.com/datasets/islo-labs/reward-hack-bench",
          "tasks": 8,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-punitarani-harvey-labs",
      "profileId": null,
      "name": "punitarani/harvey-labs",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/punitarani/harvey-labs",
      "catalogUrl": "https://hub.harborframework.com/datasets/punitarani/harvey-labs",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "punitarani/harvey-labs",
          "url": "https://hub.harborframework.com/datasets/punitarani/harvey-labs",
          "tasks": 1760,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ryanmarten-terminal-bench-4-no-prebuild",
      "profileId": null,
      "name": "ryanmarten/terminal-bench-4-no-prebuild",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ryanmarten/terminal-bench-4-no-prebuild",
      "catalogUrl": "https://hub.harborframework.com/datasets/ryanmarten/terminal-bench-4-no-prebuild",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ryanmarten/terminal-bench-4-no-prebuild",
          "url": "https://hub.harborframework.com/datasets/ryanmarten/terminal-bench-4-no-prebuild",
          "tasks": 66,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-vmax-modal-modal-port-v1-train-patched",
      "profileId": null,
      "name": "vmax-modal/modal-port-v1-train-patched",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/vmax-modal/modal-port-v1-train-patched",
      "catalogUrl": "https://hub.harborframework.com/datasets/vmax-modal/modal-port-v1-train-patched",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "vmax-modal/modal-port-v1-train-patched",
          "url": "https://hub.harborframework.com/datasets/vmax-modal/modal-port-v1-train-patched",
          "tasks": 1702,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-abundant-swe-gen-js",
      "profileId": null,
      "name": "abundant/swe-gen-js",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-js",
      "catalogUrl": "https://hub.harborframework.com/datasets/abundant/swe-gen-js",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "abundant/swe-gen-js",
          "url": "https://hub.harborframework.com/datasets/abundant/swe-gen-js",
          "tasks": 1000,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ai-forever-harness-bench-fast",
      "profileId": null,
      "name": "ai-forever/harness-bench-fast",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ai-forever/harness-bench-fast",
      "catalogUrl": "https://hub.harborframework.com/datasets/ai-forever/harness-bench-fast",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ai-forever/harness-bench-fast",
          "url": "https://hub.harborframework.com/datasets/ai-forever/harness-bench-fast",
          "tasks": 231,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-apple-mmau",
      "profileId": null,
      "name": "apple/mmau",
      "category": "Harbor catalog",
      "description": "Shortlisted family; see profile",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Relevant to a named product capability; runtime and scope caveats are documented in the benchmark profile.",
      "priority": "P3",
      "maturity": "Public detail metadata + family research",
      "sourceUrl": "https://hub.harborframework.com/datasets/apple/mmau",
      "catalogUrl": "https://hub.harborframework.com/datasets/apple/mmau",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "apple/mmau",
          "url": "https://hub.harborframework.com/datasets/apple/mmau",
          "tasks": 1000,
          "access": "Public",
          "review": "Public detail metadata + family research"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-evoeval-evoeval",
      "profileId": null,
      "name": "evoeval/evoeval",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/evoeval/evoeval",
      "catalogUrl": "https://hub.harborframework.com/datasets/evoeval/evoeval",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "evoeval/evoeval",
          "url": "https://hub.harborframework.com/datasets/evoeval/evoeval",
          "tasks": 100,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-featurebench-featurebench-lite",
      "profileId": null,
      "name": "featurebench/featurebench-lite",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench-lite",
      "catalogUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench-lite",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "featurebench/featurebench-lite",
          "url": "https://hub.harborframework.com/datasets/featurebench/featurebench-lite",
          "tasks": 30,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-infra-bench-infra-bench-v1",
      "profileId": null,
      "name": "infra-bench/infra-bench-v1",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/infra-bench/infra-bench-v1",
      "catalogUrl": "https://hub.harborframework.com/datasets/infra-bench/infra-bench-v1",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "infra-bench/infra-bench-v1",
          "url": "https://hub.harborframework.com/datasets/infra-bench/infra-bench-v1",
          "tasks": 20,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ivanleo-agent-search",
      "profileId": null,
      "name": "ivanleo/agent-search",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ivanleo/agent-search",
      "catalogUrl": "https://hub.harborframework.com/datasets/ivanleo/agent-search",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ivanleo/agent-search",
          "url": "https://hub.harborframework.com/datasets/ivanleo/agent-search",
          "tasks": 20,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-kumo-kumo-parity",
      "profileId": null,
      "name": "kumo/kumo-parity",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/kumo/kumo-parity",
      "catalogUrl": "https://hub.harborframework.com/datasets/kumo/kumo-parity",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "kumo/kumo-parity",
          "url": "https://hub.harborframework.com/datasets/kumo/kumo-parity",
          "tasks": 212,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tmax-tmax-15k-harbor-trial",
      "profileId": null,
      "name": "tmax/TMax-15K-Harbor-trial",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/tmax/TMax-15K-Harbor-trial",
      "catalogUrl": "https://hub.harborframework.com/datasets/tmax/TMax-15K-Harbor-trial",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tmax/TMax-15K-Harbor-trial",
          "url": "https://hub.harborframework.com/datasets/tmax/TMax-15K-Harbor-trial",
          "tasks": 12,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-bespokelabs-autoresearch-exam",
      "profileId": null,
      "name": "bespokelabs/autoresearch-exam",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/bespokelabs/autoresearch-exam",
      "catalogUrl": "https://hub.harborframework.com/datasets/bespokelabs/autoresearch-exam",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "bespokelabs/autoresearch-exam",
          "url": "https://hub.harborframework.com/datasets/bespokelabs/autoresearch-exam",
          "tasks": 29,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-deveval-deveval",
      "profileId": null,
      "name": "deveval/deveval",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/deveval/deveval",
      "catalogUrl": "https://hub.harborframework.com/datasets/deveval/deveval",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "deveval/deveval",
          "url": "https://hub.harborframework.com/datasets/deveval/deveval",
          "tasks": 63,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-featurebench-featurebench",
      "profileId": null,
      "name": "featurebench/featurebench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench",
      "catalogUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "featurebench/featurebench",
          "url": "https://hub.harborframework.com/datasets/featurebench/featurebench",
          "tasks": 200,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-gorilla-bfcl-parity",
      "profileId": null,
      "name": "gorilla/bfcl_parity",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/gorilla/bfcl_parity",
      "catalogUrl": "https://hub.harborframework.com/datasets/gorilla/bfcl_parity",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "gorilla/bfcl_parity",
          "url": "https://hub.harborframework.com/datasets/gorilla/bfcl_parity",
          "tasks": 123,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-litecoder-litecoder-rl",
      "profileId": null,
      "name": "LiteCoder/LiteCoder-rl",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/LiteCoder/LiteCoder-rl",
      "catalogUrl": "https://hub.harborframework.com/datasets/LiteCoder/LiteCoder-rl",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "LiteCoder/LiteCoder-rl",
          "url": "https://hub.harborframework.com/datasets/LiteCoder/LiteCoder-rl",
          "tasks": 602,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openai-swe-lancer-diamond-all",
      "profileId": null,
      "name": "openai/swe-lancer-diamond-all",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-all",
      "catalogUrl": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-all",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openai/swe-lancer-diamond-all",
          "url": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-all",
          "tasks": 463,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-pytest-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-pytest-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-pytest-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-pytest-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-pytest-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-pytest-v2",
          "tasks": 500,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-rexbench-rexbench",
      "profileId": null,
      "name": "rexbench/rexbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/rexbench/rexbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/rexbench/rexbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "rexbench/rexbench",
          "url": "https://hub.harborframework.com/datasets/rexbench/rexbench",
          "tasks": 2,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-termigen-termigen-environments",
      "profileId": null,
      "name": "termigen/termigen-environments",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/termigen/termigen-environments",
      "catalogUrl": "https://hub.harborframework.com/datasets/termigen/termigen-environments",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "termigen/termigen-environments",
          "url": "https://hub.harborframework.com/datasets/termigen/termigen-environments",
          "tasks": 3566,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ale-rsi-post-training",
      "profileId": null,
      "name": "ale/rsi-post-training",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ale/rsi-post-training",
      "catalogUrl": "https://hub.harborframework.com/datasets/ale/rsi-post-training",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ale/rsi-post-training",
          "url": "https://hub.harborframework.com/datasets/ale/rsi-post-training",
          "tasks": 6,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ashantanu-task1-v3-1-stock-research-basket-rolling",
      "profileId": null,
      "name": "ashantanu/task1_v3_1_stock_research_basket_rolling",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ashantanu/task1_v3_1_stock_research_basket_rolling",
      "catalogUrl": "https://hub.harborframework.com/datasets/ashantanu/task1_v3_1_stock_research_basket_rolling",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ashantanu/task1_v3_1_stock_research_basket_rolling",
          "url": "https://hub.harborframework.com/datasets/ashantanu/task1_v3_1_stock_research_basket_rolling",
          "tasks": 8,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-domainbench-24",
      "profileId": null,
      "name": "blobfishai/domainbench-24",
      "category": "Harbor catalog",
      "description": "House/vendor suite; separate validation needed",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/domainbench-24",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/domainbench-24",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/domainbench-24",
          "url": "https://hub.harborframework.com/datasets/blobfishai/domainbench-24",
          "tasks": 24,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-codepde-codepde",
      "profileId": null,
      "name": "codepde/codepde",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/codepde/codepde",
      "catalogUrl": "https://hub.harborframework.com/datasets/codepde/codepde",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "codepde/codepde",
          "url": "https://hub.harborframework.com/datasets/codepde/codepde",
          "tasks": 5,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-luosuu-swe-kokkos-bench",
      "profileId": null,
      "name": "luosuu/SWE-kokkos-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/luosuu/SWE-kokkos-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/luosuu/SWE-kokkos-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "luosuu/SWE-kokkos-bench",
          "url": "https://hub.harborframework.com/datasets/luosuu/SWE-kokkos-bench",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-novitaai-tb21-code-debug",
      "profileId": null,
      "name": "NovitaAI/tb21-code-debug",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-code-debug",
      "catalogUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-code-debug",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "NovitaAI/tb21-code-debug",
          "url": "https://hub.harborframework.com/datasets/NovitaAI/tb21-code-debug",
          "tasks": 32,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-novitaai-tb21-systems-security",
      "profileId": null,
      "name": "NovitaAI/tb21-systems-security",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-systems-security",
      "catalogUrl": "https://hub.harborframework.com/datasets/NovitaAI/tb21-systems-security",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "NovitaAI/tb21-systems-security",
          "url": "https://hub.harborframework.com/datasets/NovitaAI/tb21-systems-security",
          "tasks": 22,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openai-mmmlu",
      "profileId": null,
      "name": "openai/mmmlu",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/openai/mmmlu",
      "catalogUrl": "https://hub.harborframework.com/datasets/openai/mmmlu",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openai/mmmlu",
          "url": "https://hub.harborframework.com/datasets/openai/mmmlu",
          "tasks": 150,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-pymethods2test-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-pymethods2test-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pymethods2test-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pymethods2test-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-pymethods2test-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pymethods2test-v3",
          "tasks": 500,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-php-large-v6",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-php-large-v6",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-php-large-v6",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-php-large-v6",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-php-large-v6",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-php-large-v6",
          "tasks": 3789,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-vmax-vmax-tasks",
      "profileId": null,
      "name": "vmax/vmax-tasks",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/vmax/vmax-tasks",
      "catalogUrl": "https://hub.harborframework.com/datasets/vmax/vmax-tasks",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "vmax/vmax-tasks",
          "url": "https://hub.harborframework.com/datasets/vmax/vmax-tasks",
          "tasks": 1043,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-atm-bench-atm-bench-hard-sgm",
      "profileId": null,
      "name": "atm-bench/atm-bench-hard-sgm",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/atm-bench/atm-bench-hard-sgm",
      "catalogUrl": "https://hub.harborframework.com/datasets/atm-bench/atm-bench-hard-sgm",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "atm-bench/atm-bench-hard-sgm",
          "url": "https://hub.harborframework.com/datasets/atm-bench/atm-bench-hard-sgm",
          "tasks": 31,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-dealbench-100-suite",
      "profileId": null,
      "name": "DealBench-100",
      "category": "Investment banking",
      "description": "Long-horizon investment-banking execution across source control, QoE, comps, precedents, DCF, LBO, merger math, bids, model-to-deck consistency, approvals, and review-ready client handoffs.",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/dealbench-100-suite",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/dealbench-100-suite",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/dealbench-100-suite",
          "url": "https://hub.harborframework.com/datasets/blobfishai/dealbench-100-suite",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": "/benchmarks/dealbench-100",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "harbor-blobfishai-ledgerbench-100",
      "profileId": null,
      "name": "LedgerBench-100",
      "category": "Financial operations",
      "description": "Corporate-finance work across 22 families: a D365-shaped ERP, Odoo, subsidiary books, drive, email, documents, and real SEC XBRL filings behind 8 MCP servers.",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/ledgerbench-100",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/ledgerbench-100",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/ledgerbench-100",
          "url": "https://hub.harborframework.com/datasets/blobfishai/ledgerbench-100",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": "/benchmarks/ledgerbench-100",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "harbor-crustbench-crustbench",
      "profileId": null,
      "name": "crustbench/crustbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/crustbench/crustbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/crustbench/crustbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "crustbench/crustbench",
          "url": "https://hub.harborframework.com/datasets/crustbench/crustbench",
          "tasks": 100,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-hwe-bench-hwe-bench",
      "profileId": null,
      "name": "hwe-bench/hwe-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/hwe-bench/hwe-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/hwe-bench/hwe-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "hwe-bench/hwe-bench",
          "url": "https://hub.harborframework.com/datasets/hwe-bench/hwe-bench",
          "tasks": 77,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ineqmath-ineqmath",
      "profileId": null,
      "name": "ineqmath/ineqmath",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/ineqmath/ineqmath",
      "catalogUrl": "https://hub.harborframework.com/datasets/ineqmath/ineqmath",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ineqmath/ineqmath",
          "url": "https://hub.harborframework.com/datasets/ineqmath/ineqmath",
          "tasks": 100,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-islo-labs-reward-hack-bench-control",
      "profileId": null,
      "name": "islo-labs/reward-hack-bench-control",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/islo-labs/reward-hack-bench-control",
      "catalogUrl": "https://hub.harborframework.com/datasets/islo-labs/reward-hack-bench-control",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "islo-labs/reward-hack-bench-control",
          "url": "https://hub.harborframework.com/datasets/islo-labs/reward-hack-bench-control",
          "tasks": 8,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-kumo-kumo-hard",
      "profileId": null,
      "name": "kumo/kumo-hard",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/kumo/kumo-hard",
      "catalogUrl": "https://hub.harborframework.com/datasets/kumo/kumo-hard",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "kumo/kumo-hard",
          "url": "https://hub.harborframework.com/datasets/kumo/kumo-hard",
          "tasks": 250,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-lcb-longswebench-32k",
      "profileId": null,
      "name": "lcb/longswebench-32k",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/lcb/longswebench-32k",
      "catalogUrl": "https://hub.harborframework.com/datasets/lcb/longswebench-32k",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "lcb/longswebench-32k",
          "url": "https://hub.harborframework.com/datasets/lcb/longswebench-32k",
          "tasks": 3,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-minnesotanlp-aar",
      "profileId": null,
      "name": "minnesotanlp/aar",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/minnesotanlp/aar",
      "catalogUrl": "https://hub.harborframework.com/datasets/minnesotanlp/aar",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "minnesotanlp/aar",
          "url": "https://hub.harborframework.com/datasets/minnesotanlp/aar",
          "tasks": 1400,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-nvats-codeskills-bench",
      "profileId": null,
      "name": "nvats/codeskills-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/nvats/codeskills-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/nvats/codeskills-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "nvats/codeskills-bench",
          "url": "https://hub.harborframework.com/datasets/nvats/codeskills-bench",
          "tasks": 23,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-curriculum-medium",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-curriculum-medium",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-medium",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-medium",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-curriculum-medium",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-medium",
          "tasks": 512,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-e2egit-large",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-e2egit-large",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-e2egit-large",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-e2egit-large",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-e2egit-large",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-e2egit-large",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-exercism-python-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-exercism-python-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-exercism-python-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-exercism-python-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-exercism-python-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-exercism-python-v2",
          "tasks": 133,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-multifile",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-multifile",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-multifile",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-multifile",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-multifile",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-multifile",
          "tasks": 4907,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-nemotron-cpp",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-nemotron-cpp",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-nemotron-cpp",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-nemotron-cpp",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-nemotron-cpp",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-nemotron-cpp",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-pr-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-pr-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pr-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pr-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-pr-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pr-v2",
          "tasks": 4793,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-scaffold-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-scaffold-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-scaffold-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-scaffold-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-scaffold-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-scaffold-v2",
          "tasks": 4861,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-dockerfile-gpt5mini-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-dockerfile-gpt5mini-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-dockerfile-gpt5mini-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-dockerfile-gpt5mini-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-dockerfile-gpt5mini-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-dockerfile-gpt5mini-v3",
          "tasks": 4137,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-rust-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-rust-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-rust-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-rust-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-rust-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-rust-v2",
          "tasks": 9987,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-unitsyn-python-large",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-unitsyn-python-large",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-unitsyn-python-large",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-unitsyn-python-large",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-unitsyn-python-large",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-unitsyn-python-large",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h6-test-quality-top25",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h6-test-quality-top25",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h6-test-quality-top25",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h6-test-quality-top25",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h6-test-quality-top25",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h6-test-quality-top25",
          "tasks": 2747,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-agent-calendar",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-agent-calendar",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-agent-calendar",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-agent-calendar",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-agent-calendar",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-agent-calendar",
          "tasks": 3358,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-math-advanced-calculations-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-math-advanced-calculations-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-math-advanced-calculations-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-math-advanced-calculations-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-math-advanced-calculations-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-math-advanced-calculations-v3",
          "tasks": 5291,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-orinlabs-horizon-1-public",
      "profileId": null,
      "name": "orinlabs/horizon-1-public",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/orinlabs/horizon-1-public",
      "catalogUrl": "https://hub.harborframework.com/datasets/orinlabs/horizon-1-public",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "orinlabs/horizon-1-public",
          "url": "https://hub.harborframework.com/datasets/orinlabs/horizon-1-public",
          "tasks": 3,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-qcircuitbench-qcircuitbench",
      "profileId": null,
      "name": "qcircuitbench/qcircuitbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/qcircuitbench/qcircuitbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/qcircuitbench/qcircuitbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "qcircuitbench/qcircuitbench",
          "url": "https://hub.harborframework.com/datasets/qcircuitbench/qcircuitbench",
          "tasks": 28,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-rsi-index-post-training",
      "profileId": null,
      "name": "rsi-index/post-training",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/rsi-index/post-training",
      "catalogUrl": "https://hub.harborframework.com/datasets/rsi-index/post-training",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "rsi-index/post-training",
          "url": "https://hub.harborframework.com/datasets/rsi-index/post-training",
          "tasks": 1,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-satbench-satbench",
      "profileId": null,
      "name": "satbench/satbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/satbench/satbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/satbench/satbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "satbench/satbench",
          "url": "https://hub.harborframework.com/datasets/satbench/satbench",
          "tasks": 2100,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-svgap-svgap-reset-release",
      "profileId": null,
      "name": "svgap/svgap-reset-release",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/svgap/svgap-reset-release",
      "catalogUrl": "https://hub.harborframework.com/datasets/svgap/svgap-reset-release",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "svgap/svgap-reset-release",
          "url": "https://hub.harborframework.com/datasets/svgap/svgap-reset-release",
          "tasks": 8,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-swt-bench-swt-bench-verified",
      "profileId": null,
      "name": "swt-bench/swt-bench-verified",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/swt-bench/swt-bench-verified",
      "catalogUrl": "https://hub.harborframework.com/datasets/swt-bench/swt-bench-verified",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "swt-bench/swt-bench-verified",
          "url": "https://hub.harborframework.com/datasets/swt-bench/swt-bench-verified",
          "tasks": 433,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-thetalab-vector-edit-gym",
      "profileId": null,
      "name": "thetalab/vector-edit-gym",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/thetalab/vector-edit-gym",
      "catalogUrl": "https://hub.harborframework.com/datasets/thetalab/vector-edit-gym",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "thetalab/vector-edit-gym",
          "url": "https://hub.harborframework.com/datasets/thetalab/vector-edit-gym",
          "tasks": 106,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-usaco-usaco",
      "profileId": null,
      "name": "usaco/usaco",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/usaco/usaco",
      "catalogUrl": "https://hub.harborframework.com/datasets/usaco/usaco",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "usaco/usaco",
          "url": "https://hub.harborframework.com/datasets/usaco/usaco",
          "tasks": 304,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-abhishek203-task-aw",
      "profileId": null,
      "name": "abhishek203/task-aw",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/abhishek203/task-aw",
      "catalogUrl": "https://hub.harborframework.com/datasets/abhishek203/task-aw",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "abhishek203/task-aw",
          "url": "https://hub.harborframework.com/datasets/abhishek203/task-aw",
          "tasks": 5,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-aryaniyaps-lamina-bench",
      "profileId": null,
      "name": "aryaniyaps/lamina-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/aryaniyaps/lamina-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/aryaniyaps/lamina-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "aryaniyaps/lamina-bench",
          "url": "https://hub.harborframework.com/datasets/aryaniyaps/lamina-bench",
          "tasks": 20,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-erpbench-100-suite",
      "profileId": null,
      "name": "ERPBench-100",
      "category": "Enterprise resource planning",
      "description": "Long-horizon ERP execution across order import, shipment verification, receipt application, reorder requisitions, receiving and three-way match, worker document compliance, shift rollups, channel-order sync, hiring within quota, and effective-dated price batches.",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/erpbench-100-suite",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/erpbench-100-suite",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/erpbench-100-suite",
          "url": "https://hub.harborframework.com/datasets/blobfishai/erpbench-100-suite",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": "/benchmarks/erpbench-100",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "harbor-featurebench-featurebench-lite-modal",
      "profileId": null,
      "name": "featurebench/featurebench-lite-modal",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench-lite-modal",
      "catalogUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench-lite-modal",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "featurebench/featurebench-lite-modal",
          "url": "https://hub.harborframework.com/datasets/featurebench/featurebench-lite-modal",
          "tasks": 30,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-itmo-autods-scientific-cases",
      "profileId": null,
      "name": "itmo-autods/scientific-cases",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/itmo-autods/scientific-cases",
      "catalogUrl": "https://hub.harborframework.com/datasets/itmo-autods/scientific-cases",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "itmo-autods/scientific-cases",
          "url": "https://hub.harborframework.com/datasets/itmo-autods/scientific-cases",
          "tasks": 3,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-lamina-product-bench",
      "profileId": null,
      "name": "lamina/product-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/lamina/product-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/lamina/product-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "lamina/product-bench",
          "url": "https://hub.harborframework.com/datasets/lamina/product-bench",
          "tasks": 12,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-michaely310-devopsgym",
      "profileId": null,
      "name": "MichaelY310/devopsgym",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/MichaelY310/devopsgym",
      "catalogUrl": "https://hub.harborframework.com/datasets/MichaelY310/devopsgym",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "MichaelY310/devopsgym",
          "url": "https://hub.harborframework.com/datasets/MichaelY310/devopsgym",
          "tasks": 728,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openai-swe-lancer-diamond-ic",
      "profileId": null,
      "name": "openai/swe-lancer-diamond-ic",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-ic",
      "catalogUrl": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-ic",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openai/swe-lancer-diamond-ic",
          "url": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-ic",
          "tasks": 198,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-curriculum-easy",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-curriculum-easy",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-easy",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-easy",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-curriculum-easy",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-easy",
          "tasks": 514,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-curriculum-hard",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-curriculum-hard",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-hard",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-hard",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-curriculum-hard",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-curriculum-hard",
          "tasks": 506,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-ghactions-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-ghactions-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-ghactions-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-ghactions-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-ghactions-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-ghactions-v3",
          "tasks": 9930,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-issue",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-issue",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-issue",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-issue",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-issue",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-issue",
          "tasks": 4830,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-nemotron-junit",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-nemotron-junit",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-nemotron-junit",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-nemotron-junit",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-nemotron-junit",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-nemotron-junit",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-bash-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-bash-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-bash-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-v3",
          "tasks": 9384,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-go-v4",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-go-v4",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-go-v4",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-go-v4",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-go-v4",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-go-v4",
          "tasks": 2313,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-jest-large",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-jest-large",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-jest-large",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-jest-large",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-jest-large",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-jest-large",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-ruby-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-ruby-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-ruby-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-ruby-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-ruby-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-ruby-v2",
          "tasks": 8627,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-nano-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-nano-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-nano-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-nano-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-nano-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-nano-v2",
          "tasks": 9999,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-baseline-uniform-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-baseline-uniform-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-baseline-uniform-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-baseline-uniform-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-baseline-uniform-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-baseline-uniform-v2",
          "tasks": 3718,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-instruction-following-adversarial-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-adversarial-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-adversarial-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-adversarial-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-adversarial-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-adversarial-v3",
          "tasks": 1000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-swegym-tasks-patched-validated-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-swegym-tasks-patched-validated-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swegym-tasks-patched-validated-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swegym-tasks-patched-validated-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-swegym-tasks-patched-validated-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swegym-tasks-patched-validated-v2",
          "tasks": 989,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-polyvorlabs-cyberdefense-bench",
      "profileId": null,
      "name": "polyvorlabs/cyberdefense-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/polyvorlabs/cyberdefense-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/polyvorlabs/cyberdefense-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "polyvorlabs/cyberdefense-bench",
          "url": "https://hub.harborframework.com/datasets/polyvorlabs/cyberdefense-bench",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-red-hat-ai-swe-benchify-hard",
      "profileId": null,
      "name": "red-hat-ai/SWE-benchify-hard",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/red-hat-ai/SWE-benchify-hard",
      "catalogUrl": "https://hub.harborframework.com/datasets/red-hat-ai/SWE-benchify-hard",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "red-hat-ai/SWE-benchify-hard",
          "url": "https://hub.harborframework.com/datasets/red-hat-ai/SWE-benchify-hard",
          "tasks": 284,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ryanmarten-tb-test",
      "profileId": null,
      "name": "ryanmarten/tb-test",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ryanmarten/tb-test",
      "catalogUrl": "https://hub.harborframework.com/datasets/ryanmarten/tb-test",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ryanmarten/tb-test",
          "url": "https://hub.harborframework.com/datasets/ryanmarten/tb-test",
          "tasks": 1,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-shiv-eshwar-lb6-dev-pilot-issue18-rewardkit",
      "profileId": null,
      "name": "shiv-eshwar/lb6-dev-pilot-issue18-rewardkit",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/shiv-eshwar/lb6-dev-pilot-issue18-rewardkit",
      "catalogUrl": "https://hub.harborframework.com/datasets/shiv-eshwar/lb6-dev-pilot-issue18-rewardkit",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "shiv-eshwar/lb6-dev-pilot-issue18-rewardkit",
          "url": "https://hub.harborframework.com/datasets/shiv-eshwar/lb6-dev-pilot-issue18-rewardkit",
          "tasks": 12,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-xiaoboai-pawbench",
      "profileId": null,
      "name": "xiaoboai/pawbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/xiaoboai/pawbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/xiaoboai/pawbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "xiaoboai/pawbench",
          "url": "https://hub.harborframework.com/datasets/xiaoboai/pawbench",
          "tasks": 150,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-xlang-ds-1000",
      "profileId": null,
      "name": "xlang/ds-1000",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/xlang/ds-1000",
      "catalogUrl": "https://hub.harborframework.com/datasets/xlang/ds-1000",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "xlang/ds-1000",
          "url": "https://hub.harborframework.com/datasets/xlang/ds-1000",
          "tasks": 1000,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-alibaba-aaig-alibaba-ctf",
      "profileId": null,
      "name": "alibaba-aaig/alibaba-ctf",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/alibaba-aaig/alibaba-ctf",
      "catalogUrl": "https://hub.harborframework.com/datasets/alibaba-aaig/alibaba-ctf",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "alibaba-aaig/alibaba-ctf",
          "url": "https://hub.harborframework.com/datasets/alibaba-aaig/alibaba-ctf",
          "tasks": 87,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-camel-ai-seta-env",
      "profileId": null,
      "name": "camel-ai/seta-env",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/camel-ai/seta-env",
      "catalogUrl": "https://hub.harborframework.com/datasets/camel-ai/seta-env",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "camel-ai/seta-env",
          "url": "https://hub.harborframework.com/datasets/camel-ai/seta-env",
          "tasks": 1376,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-cmu-refav",
      "profileId": null,
      "name": "cmu/refav",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/cmu/refav",
      "catalogUrl": "https://hub.harborframework.com/datasets/cmu/refav",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "cmu/refav",
          "url": "https://hub.harborframework.com/datasets/cmu/refav",
          "tasks": 1500,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-cookbook-test",
      "profileId": null,
      "name": "cookbook/test",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/cookbook/test",
      "catalogUrl": "https://hub.harborframework.com/datasets/cookbook/test",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "cookbook/test",
          "url": "https://hub.harborframework.com/datasets/cookbook/test",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-featurebench-featurebench-modal",
      "profileId": null,
      "name": "featurebench/featurebench-modal",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench-modal",
      "catalogUrl": "https://hub.harborframework.com/datasets/featurebench/featurebench-modal",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "featurebench/featurebench-modal",
          "url": "https://hub.harborframework.com/datasets/featurebench/featurebench-modal",
          "tasks": 200,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-hypernymai-hypernym-swebench-artifact-replay",
      "profileId": null,
      "name": "hypernymai/hypernym-swebench-artifact-replay",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/hypernymai/hypernym-swebench-artifact-replay",
      "catalogUrl": "https://hub.harborframework.com/datasets/hypernymai/hypernym-swebench-artifact-replay",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "hypernymai/hypernym-swebench-artifact-replay",
          "url": "https://hub.harborframework.com/datasets/hypernymai/hypernym-swebench-artifact-replay",
          "tasks": 8,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-kumo-kumo-1",
      "profileId": null,
      "name": "kumo/kumo-1",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/kumo/kumo-1",
      "catalogUrl": "https://hub.harborframework.com/datasets/kumo/kumo-1",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "kumo/kumo-1",
          "url": "https://hub.harborframework.com/datasets/kumo/kumo-1",
          "tasks": 5300,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-kumo-kumo-easy",
      "profileId": null,
      "name": "kumo/kumo-easy",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/kumo/kumo-easy",
      "catalogUrl": "https://hub.harborframework.com/datasets/kumo/kumo-easy",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "kumo/kumo-easy",
          "url": "https://hub.harborframework.com/datasets/kumo/kumo-easy",
          "tasks": 5050,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-lawbench-lawbench",
      "profileId": null,
      "name": "lawbench/lawbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/lawbench/lawbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/lawbench/lawbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "lawbench/lawbench",
          "url": "https://hub.harborframework.com/datasets/lawbench/lawbench",
          "tasks": 1000,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openai-swe-lancer-diamond-manager",
      "profileId": null,
      "name": "openai/swe-lancer-diamond-manager",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-manager",
      "catalogUrl": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-manager",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openai/swe-lancer-diamond-manager",
          "url": "https://hub.harborframework.com/datasets/openai/swe-lancer-diamond-manager",
          "tasks": 265,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-bash-withtests-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-bash-withtests-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-withtests-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-withtests-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-bash-withtests-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-withtests-v2",
          "tasks": 8922,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-csharp-v5",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-csharp-v5",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-csharp-v5",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-csharp-v5",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-csharp-v5",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-csharp-v5",
          "tasks": 9485,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nl2bash-tasks-cleaned-oracle",
      "profileId": null,
      "name": "openthoughts/tasktrove-nl2bash-tasks-cleaned-oracle",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nl2bash-tasks-cleaned-oracle",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nl2bash-tasks-cleaned-oracle",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nl2bash-tasks-cleaned-oracle",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nl2bash-tasks-cleaned-oracle",
          "tasks": 1570,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-openswe-tasks-patched-v5-oracle-success",
      "profileId": null,
      "name": "openthoughts/tasktrove-openswe-tasks-patched-v5-oracle-success",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-openswe-tasks-patched-v5-oracle-success",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-openswe-tasks-patched-v5-oracle-success",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-openswe-tasks-patched-v5-oracle-success",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-openswe-tasks-patched-v5-oracle-success",
          "tasks": 17504,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-swe-rebench-patched-oracle",
      "profileId": null,
      "name": "openthoughts/tasktrove-swe-rebench-patched-oracle",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swe-rebench-patched-oracle",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swe-rebench-patched-oracle",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-swe-rebench-patched-oracle",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swe-rebench-patched-oracle",
          "tasks": 3787,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-polyvorlabs-revbench",
      "profileId": null,
      "name": "polyvorlabs/revbench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/polyvorlabs/revbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/polyvorlabs/revbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "polyvorlabs/revbench",
          "url": "https://hub.harborframework.com/datasets/polyvorlabs/revbench",
          "tasks": 22,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-gustofied-compute-bazaar-bench",
      "profileId": null,
      "name": "gustofied/compute-bazaar-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/gustofied/compute-bazaar-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/gustofied/compute-bazaar-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "gustofied/compute-bazaar-bench",
          "url": "https://hub.harborframework.com/datasets/gustofied/compute-bazaar-bench",
          "tasks": 4,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-itmo-autods-openpoly-tg-case",
      "profileId": null,
      "name": "itmo-autods/openpoly-tg-case",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/itmo-autods/openpoly-tg-case",
      "catalogUrl": "https://hub.harborframework.com/datasets/itmo-autods/openpoly-tg-case",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "itmo-autods/openpoly-tg-case",
          "url": "https://hub.harborframework.com/datasets/itmo-autods/openpoly-tg-case",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rle-detailed-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rle-detailed-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-detailed-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-detailed-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rle-detailed-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-detailed-v3",
          "tasks": 413,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rle-error-report-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rle-error-report-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-error-report-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-error-report-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rle-error-report-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-error-report-v3",
          "tasks": 261,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rle-github-issue-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rle-github-issue-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-github-issue-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-github-issue-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rle-github-issue-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-github-issue-v3",
          "tasks": 264,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rle-heavy-padding-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rle-heavy-padding-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-heavy-padding-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-heavy-padding-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rle-heavy-padding-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-heavy-padding-v2",
          "tasks": 784,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rle-minimal-instructions-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rle-minimal-instructions-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-minimal-instructions-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-minimal-instructions-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rle-minimal-instructions-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-minimal-instructions-v3",
          "tasks": 233,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-e2egit-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-e2egit-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-e2egit-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-e2egit-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-e2egit-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-e2egit-v2",
          "tasks": 500,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-dockerfile-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-dockerfile-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-dockerfile-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-dockerfile-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-dockerfile-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-dockerfile-v2",
          "tasks": 497,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-jest-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-jest-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-jest-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-jest-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-jest-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-jest-v2",
          "tasks": 500,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-junit-v6",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-junit-v6",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-junit-v6",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-junit-v6",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-junit-v6",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-junit-v6",
          "tasks": 872,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-php-v2-v6",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-php-v2-v6",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-php-v2-v6",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-php-v2-v6",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-php-v2-v6",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-php-v2-v6",
          "tasks": 438,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-pytest-large",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-pytest-large",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-pytest-large",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-pytest-large",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-pytest-large",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-pytest-large",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-unitsyn-python-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-unitsyn-python-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-unitsyn-python-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-unitsyn-python-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-unitsyn-python-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-unitsyn-python-v3",
          "tasks": 500,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h10-reward-staged-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h10-reward-staged-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-staged-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-staged-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h10-reward-staged-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-staged-v2",
          "tasks": 3873,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h11-single-skill-only-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h11-single-skill-only-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h11-single-skill-only-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h11-single-skill-only-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h11-single-skill-only-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h11-single-skill-only-v2",
          "tasks": 2873,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h2-language-proportional",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h2-language-proportional",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h2-language-proportional",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h2-language-proportional",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h2-language-proportional",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h2-language-proportional",
          "tasks": 4135,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h4-binary-easy",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h4-binary-easy",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h4-binary-easy",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h4-binary-easy",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h4-binary-easy",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h4-binary-easy",
          "tasks": 2010,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h5-skill-diverse-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h5-skill-diverse-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h5-skill-diverse-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h5-skill-diverse-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h5-skill-diverse-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h5-skill-diverse-v2",
          "tasks": 3166,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h8-adversarial-tests-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h8-adversarial-tests-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h8-adversarial-tests-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h8-adversarial-tests-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h8-adversarial-tests-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h8-adversarial-tests-v2",
          "tasks": 2873,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h8-original-tests-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h8-original-tests-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h8-original-tests-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h8-original-tests-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h8-original-tests-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h8-original-tests-v2",
          "tasks": 2862,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-agent-workplace-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-agent-workplace-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-agent-workplace-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-agent-workplace-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-agent-workplace-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-agent-workplace-v2",
          "tasks": 297,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-competitive-coding",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-competitive-coding",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-competitive-coding",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-competitive-coding",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-competitive-coding",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-competitive-coding",
          "tasks": 15713,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-instruction-following-structured",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-structured",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-structured",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-structured",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-structured",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-structured",
          "tasks": 9437,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-knowledge-web-search-mcqa",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-knowledge-web-search-mcqa",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-knowledge-web-search-mcqa",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-knowledge-web-search-mcqa",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-knowledge-web-search-mcqa",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-knowledge-web-search-mcqa",
          "tasks": 2915,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-r2egym-patched-full-oracle",
      "profileId": null,
      "name": "openthoughts/tasktrove-r2egym-patched-full-oracle",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-r2egym-patched-full-oracle",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-r2egym-patched-full-oracle",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-r2egym-patched-full-oracle",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-r2egym-patched-full-oracle",
          "tasks": 3328,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-swe-rebench-v2-patched-oracle",
      "profileId": null,
      "name": "openthoughts/tasktrove-swe-rebench-v2-patched-oracle",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swe-rebench-v2-patched-oracle",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swe-rebench-v2-patched-oracle",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-swe-rebench-v2-patched-oracle",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swe-rebench-v2-patched-oracle",
          "tasks": 18341,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-swesmith-oracle-filtered",
      "profileId": null,
      "name": "openthoughts/tasktrove-swesmith-oracle-filtered",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swesmith-oracle-filtered",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swesmith-oracle-filtered",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-swesmith-oracle-filtered",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-swesmith-oracle-filtered",
          "tasks": 12942,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-ryanmarten-tb4-preview",
      "profileId": null,
      "name": "ryanmarten/tb4-preview",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/ryanmarten/tb4-preview",
      "catalogUrl": "https://hub.harborframework.com/datasets/ryanmarten/tb4-preview",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "ryanmarten/tb4-preview",
          "url": "https://hub.harborframework.com/datasets/ryanmarten/tb4-preview",
          "tasks": 66,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-titouanlne-dbt-labs-k8s",
      "profileId": null,
      "name": "titouanlne/dbt-labs-k8s",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/titouanlne/dbt-labs-k8s",
      "catalogUrl": "https://hub.harborframework.com/datasets/titouanlne/dbt-labs-k8s",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "titouanlne/dbt-labs-k8s",
          "url": "https://hub.harborframework.com/datasets/titouanlne/dbt-labs-k8s",
          "tasks": 48,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-adameubanks-rating-generation-consistency",
      "profileId": null,
      "name": "adameubanks/rating-generation-consistency",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/adameubanks/rating-generation-consistency",
      "catalogUrl": "https://hub.harborframework.com/datasets/adameubanks/rating-generation-consistency",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "adameubanks/rating-generation-consistency",
          "url": "https://hub.harborframework.com/datasets/adameubanks/rating-generation-consistency",
          "tasks": 45,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-litigation",
      "profileId": null,
      "name": "blobfishai/litigation",
      "category": "Harbor catalog",
      "description": "House/vendor suite; separate validation needed",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/litigation",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/litigation",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/litigation",
          "url": "https://hub.harborframework.com/datasets/blobfishai/litigation",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-fritzprix-libragent-diverse-9",
      "profileId": null,
      "name": "fritzprix/libragent-diverse-9",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/fritzprix/libragent-diverse-9",
      "catalogUrl": "https://hub.harborframework.com/datasets/fritzprix/libragent-diverse-9",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "fritzprix/libragent-diverse-9",
          "url": "https://hub.harborframework.com/datasets/fritzprix/libragent-diverse-9",
          "tasks": 9,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-grandsmile-unicode",
      "profileId": null,
      "name": "grandsmile/unicode",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/grandsmile/unicode",
      "catalogUrl": "https://hub.harborframework.com/datasets/grandsmile/unicode",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "grandsmile/unicode",
          "url": "https://hub.harborframework.com/datasets/grandsmile/unicode",
          "tasks": 399,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-itmo-autods-fdata-exit-case",
      "profileId": null,
      "name": "itmo-autods/fdata-exit-case",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/itmo-autods/fdata-exit-case",
      "catalogUrl": "https://hub.harborframework.com/datasets/itmo-autods/fdata-exit-case",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "itmo-autods/fdata-exit-case",
          "url": "https://hub.harborframework.com/datasets/itmo-autods/fdata-exit-case",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-itmo-autods-maize-yield-case",
      "profileId": null,
      "name": "itmo-autods/maize-yield-case",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/itmo-autods/maize-yield-case",
      "catalogUrl": "https://hub.harborframework.com/datasets/itmo-autods/maize-yield-case",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "itmo-autods/maize-yield-case",
          "url": "https://hub.harborframework.com/datasets/itmo-autods/maize-yield-case",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-lica-world-gdb",
      "profileId": null,
      "name": "lica-world/gdb",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/lica-world/gdb",
      "catalogUrl": "https://hub.harborframework.com/datasets/lica-world/gdb",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "lica-world/gdb",
          "url": "https://hub.harborframework.com/datasets/lica-world/gdb",
          "tasks": 33786,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-nanoswe-swe-bench-verified",
      "profileId": null,
      "name": "nanoswe/swe-bench-verified",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/nanoswe/swe-bench-verified",
      "catalogUrl": "https://hub.harborframework.com/datasets/nanoswe/swe-bench-verified",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "nanoswe/swe-bench-verified",
          "url": "https://hub.harborframework.com/datasets/nanoswe/swe-bench-verified",
          "tasks": 1,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-open-design-open-design",
      "profileId": null,
      "name": "open-design/open-design",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/open-design/open-design",
      "catalogUrl": "https://hub.harborframework.com/datasets/open-design/open-design",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "open-design/open-design",
          "url": "https://hub.harborframework.com/datasets/open-design/open-design",
          "tasks": 7,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-code-contests-noblock",
      "profileId": null,
      "name": "openthoughts/tasktrove-code-contests-noblock",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-code-contests-noblock",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-code-contests-noblock",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-code-contests-noblock",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-code-contests-noblock",
          "tasks": 8728,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-flat25-pseudocode-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-flat25-pseudocode-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-pseudocode-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-pseudocode-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-flat25-pseudocode-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-pseudocode-v2",
          "tasks": 728,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-flat25-speed-bonus-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-flat25-speed-bonus-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-speed-bonus-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-speed-bonus-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-flat25-speed-bonus-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-speed-bonus-v2",
          "tasks": 764,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-flat25-stackoverflow-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-flat25-stackoverflow-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-stackoverflow-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-stackoverflow-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-flat25-stackoverflow-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-stackoverflow-v2",
          "tasks": 765,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-flat25-subtle-debug-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-flat25-subtle-debug-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-subtle-debug-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-subtle-debug-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-flat25-subtle-debug-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-flat25-subtle-debug-v3",
          "tasks": 289,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rle-adversarial",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rle-adversarial",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-adversarial",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-adversarial",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rle-adversarial",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rle-adversarial",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-codenet-python-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-codenet-python-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-codenet-python-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-codenet-python-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-codenet-python-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-codenet-python-v2",
          "tasks": 10000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-crosscodeeval-csharp-v4",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-crosscodeeval-csharp-v4",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-crosscodeeval-csharp-v4",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-crosscodeeval-csharp-v4",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-crosscodeeval-csharp-v4",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-crosscodeeval-csharp-v4",
          "tasks": 1768,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-crosscodeeval-java",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-crosscodeeval-java",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-crosscodeeval-java",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-crosscodeeval-java",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-crosscodeeval-java",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-crosscodeeval-java",
          "tasks": 2139,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-defects4j-v3-v4",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-defects4j-v3-v4",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-defects4j-v3-v4",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-defects4j-v3-v4",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-defects4j-v3-v4",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-defects4j-v3-v4",
          "tasks": 216,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-methods2test-large-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-methods2test-large-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-methods2test-large-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-methods2test-large-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-methods2test-large-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-methods2test-large-v2",
          "tasks": 4472,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-methods2test-large-v3",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-methods2test-large-v3",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-methods2test-large-v3",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-methods2test-large-v3",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-methods2test-large-v3",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-methods2test-large-v3",
          "tasks": 4472,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-pr",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-pr",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pr",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pr",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-pr",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pr",
          "tasks": 4793,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-taco-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-taco-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-taco-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-taco-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-taco-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-taco-v2",
          "tasks": 10000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-mini-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-mini-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-mini-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-mini-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-mini-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-freelancer-projects-sandboxes-ta-rl-gpt-5-mini-v2",
          "tasks": 9999,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-inferredbugs-sandboxes-verifier",
      "profileId": null,
      "name": "openthoughts/tasktrove-inferredbugs-sandboxes-verifier",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-inferredbugs-sandboxes-verifier",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-inferredbugs-sandboxes-verifier",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-inferredbugs-sandboxes-verifier",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-inferredbugs-sandboxes-verifier",
          "tasks": 9991,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-llm-verifier-freelancer",
      "profileId": null,
      "name": "openthoughts/tasktrove-llm-verifier-freelancer",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-llm-verifier-freelancer",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-llm-verifier-freelancer",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-llm-verifier-freelancer",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-llm-verifier-freelancer",
          "tasks": 10000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h1-struggle-zone-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h1-struggle-zone-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h1-struggle-zone-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h1-struggle-zone-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h1-struggle-zone-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h1-struggle-zone-v2",
          "tasks": 3116,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h10-reward-binary-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h10-reward-binary-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-binary-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-binary-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h10-reward-binary-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-binary-v2",
          "tasks": 2862,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h10-reward-proportional-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h10-reward-proportional-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-proportional-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-proportional-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h10-reward-proportional-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h10-reward-proportional-v2",
          "tasks": 2873,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h11-compositional-gradient-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h11-compositional-gradient-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h11-compositional-gradient-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h11-compositional-gradient-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h11-compositional-gradient-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h11-compositional-gradient-v2",
          "tasks": 3873,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h2-language-balanced-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h2-language-balanced-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h2-language-balanced-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h2-language-balanced-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h2-language-balanced-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h2-language-balanced-v2",
          "tasks": 4506,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-mix-h7-raw-volume-5k-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-mix-h7-raw-volume-5k-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h7-raw-volume-5k-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h7-raw-volume-5k-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-mix-h7-raw-volume-5k-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-mix-h7-raw-volume-5k-v2",
          "tasks": 3718,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-identity-following-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-identity-following-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-identity-following-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-identity-following-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-identity-following-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-identity-following-v2",
          "tasks": 21660,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-instruction-following-calendar",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-calendar",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-calendar",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-calendar",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-calendar",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-calendar",
          "tasks": 8387,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-titouanlne-arc-agi-2-k8s",
      "profileId": null,
      "name": "titouanlne/arc-agi-2-k8s",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/titouanlne/arc-agi-2-k8s",
      "catalogUrl": "https://hub.harborframework.com/datasets/titouanlne/arc-agi-2-k8s",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "titouanlne/arc-agi-2-k8s",
          "url": "https://hub.harborframework.com/datasets/titouanlne/arc-agi-2-k8s",
          "tasks": 167,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tomseimandi-ade-bench",
      "profileId": null,
      "name": "tomseimandi/ade-bench",
      "category": "Harbor catalog",
      "description": "Version/port/subset; avoid double counting",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Related to another benchmark family or deployment variant; establish exact membership and evaluator parity before using scores.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tomseimandi/ade-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/tomseimandi/ade-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tomseimandi/ade-bench",
          "url": "https://hub.harborframework.com/datasets/tomseimandi/ade-bench",
          "tasks": 48,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-josancamon19-physician-bench",
      "profileId": null,
      "name": "josancamon19/physician-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/josancamon19/physician-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/josancamon19/physician-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "josancamon19/physician-bench",
          "url": "https://hub.harborframework.com/datasets/josancamon19/physician-bench",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-pymethods2test-large",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-pymethods2test-large",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pymethods2test-large",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pymethods2test-large",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-pymethods2test-large",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-pymethods2test-large",
          "tasks": 5000,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-exp-rpt-stack-bash-withtests-gpt5mini-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-exp-rpt-stack-bash-withtests-gpt5mini-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-withtests-gpt5mini-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-withtests-gpt5mini-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-exp-rpt-stack-bash-withtests-gpt5mini-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-exp-rpt-stack-bash-withtests-gpt5mini-v2",
          "tasks": 8922,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-code-oracle-filtered",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-code-oracle-filtered",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-code-oracle-filtered",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-code-oracle-filtered",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-code-oracle-filtered",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-code-oracle-filtered",
          "tasks": 15165,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-nemotron-gym-instruction-following-v2",
      "profileId": null,
      "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-v2",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-v2",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-v2",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-nemotron-gym-instruction-following-v2",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-nemotron-gym-instruction-following-v2",
          "tasks": 46391,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-openthoughts-tasktrove-selfinstruct-naive-sandboxes-2-verified",
      "profileId": null,
      "name": "openthoughts/tasktrove-selfinstruct-naive-sandboxes-2-verified",
      "category": "Harbor catalog",
      "description": "Training/development/subset candidate",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Name/card indicates a training, generated, development or subset collection; inspect held-out protocol before using it as an independent evaluation.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-selfinstruct-naive-sandboxes-2-verified",
      "catalogUrl": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-selfinstruct-naive-sandboxes-2-verified",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "openthoughts/tasktrove-selfinstruct-naive-sandboxes-2-verified",
          "url": "https://hub.harborframework.com/datasets/openthoughts/tasktrove-selfinstruct-naive-sandboxes-2-verified",
          "tasks": 9638,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-orca-bench-orca-bench-verified",
      "profileId": null,
      "name": "orca-bench/orca-bench-verified",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/orca-bench/orca-bench-verified",
      "catalogUrl": "https://hub.harborframework.com/datasets/orca-bench/orca-bench-verified",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "orca-bench/orca-bench-verified",
          "url": "https://hub.harborframework.com/datasets/orca-bench/orca-bench-verified",
          "tasks": 40,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-polyvorlabs-klyrune-bench",
      "profileId": null,
      "name": "polyvorlabs/klyrune-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/polyvorlabs/klyrune-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/polyvorlabs/klyrune-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "polyvorlabs/klyrune-bench",
          "url": "https://hub.harborframework.com/datasets/polyvorlabs/klyrune-bench",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-arc-crm-6",
      "profileId": null,
      "name": "blobfishai/arc-crm-6",
      "category": "Harbor catalog",
      "description": "House/vendor suite; separate validation needed",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/arc-crm-6",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/arc-crm-6",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/arc-crm-6",
          "url": "https://hub.harborframework.com/datasets/blobfishai/arc-crm-6",
          "tasks": 6,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-hubbench",
      "profileId": null,
      "name": "blobfishai/hubbench",
      "category": "Harbor catalog",
      "description": "House/vendor suite; separate validation needed",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/hubbench",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/hubbench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/hubbench",
          "url": "https://hub.harborframework.com/datasets/blobfishai/hubbench",
          "tasks": 104,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-blobfishai-semikongbench-100",
      "profileId": null,
      "name": "SemiKongBench-100",
      "category": "Semiconductor operations",
      "description": "Long-horizon semiconductor exception work across lot genealogy, plasma etch, CMP, lithography, maintenance, materials, final test, reliability, FA/CAPA, and supply traceability.",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Listed in the catalog; not treated as independently established or famous. Catalog task counts do not prove validated coverage.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/blobfishai/semikongbench-100",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/semikongbench-100",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "blobfishai/semikongbench-100",
          "url": "https://hub.harborframework.com/datasets/blobfishai/semikongbench-100",
          "tasks": 100,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": "/benchmarks/semikongbench-100",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "harbor-nlile-registry-history-20260908",
      "profileId": null,
      "name": "nlile/registry-history-20260908",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/nlile/registry-history-20260908",
      "catalogUrl": "https://hub.harborframework.com/datasets/nlile/registry-history-20260908",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "nlile/registry-history-20260908",
          "url": "https://hub.harborframework.com/datasets/nlile/registry-history-20260908",
          "tasks": 1,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-zenml-zenml-bench",
      "profileId": null,
      "name": "zenml/zenml-bench",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Public description screening",
      "sourceUrl": "https://hub.harborframework.com/datasets/zenml/zenml-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/zenml/zenml-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "zenml/zenml-bench",
          "url": "https://hub.harborframework.com/datasets/zenml/zenml-bench",
          "tasks": 18,
          "access": "Public",
          "review": "Public description screening"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-dw-a-dw-1784073139-2608908-dw-dataset",
      "profileId": null,
      "name": "tbench-dw-a-dw_1784073139_2608908/dw-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-dw-a-dw_1784073139_2608908/dw-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-dw-a-dw_1784073139_2608908/dw-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-dw-a-dw_1784073139_2608908/dw-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-dw-a-dw_1784073139_2608908/dw-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-dw2-a-dw2-1784073189-2619080-dw2-dataset",
      "profileId": null,
      "name": "tbench-dw2-a-dw2_1784073189_2619080/dw2-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-dw2-a-dw2_1784073189_2619080/dw2-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-dw2-a-dw2_1784073189_2619080/dw2-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-dw2-a-dw2_1784073189_2619080/dw2-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-dw2-a-dw2_1784073189_2619080/dw2-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-dw3-a-dw3-1784073253-2627946-dw3-dataset",
      "profileId": null,
      "name": "tbench-dw3-a-dw3_1784073253_2627946/dw3-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-dw3-a-dw3_1784073253_2627946/dw3-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-dw3-a-dw3_1784073253_2627946/dw3-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-dw3-a-dw3_1784073253_2627946/dw3-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-dw3-a-dw3_1784073253_2627946/dw3-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-lbtest-a-xt-1784071893-2521724-xt-dataset",
      "profileId": null,
      "name": "tbench-lbtest-a-xt_1784071893_2521724/xt-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-lbtest-a-xt_1784071893_2521724/xt-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-lbtest-a-xt_1784071893_2521724/xt-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-lbtest-a-xt_1784071893_2521724/xt-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-lbtest-a-xt_1784071893_2521724/xt-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-lbtest-lb-1784071796-2517183-lbtest-dataset",
      "profileId": null,
      "name": "tbench-lbtest-lb_1784071796_2517183/lbtest-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-lbtest-lb_1784071796_2517183/lbtest-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-lbtest-lb_1784071796_2517183/lbtest-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-lbtest-lb_1784071796_2517183/lbtest-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-lbtest-lb_1784071796_2517183/lbtest-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-pubcrack-a-pc-1784073251-pubcrack-dataset",
      "profileId": null,
      "name": "tbench-pubcrack-a-pc_1784073251/pubcrack-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-pubcrack-a-pc_1784073251/pubcrack-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-pubcrack-a-pc_1784073251/pubcrack-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-pubcrack-a-pc_1784073251/pubcrack-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-pubcrack-a-pc_1784073251/pubcrack-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-rv-a-rv-1784073719-2926154-rv-dataset",
      "profileId": null,
      "name": "tbench-rv-a-rv_1784073719_2926154/rv-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-rv-a-rv_1784073719_2926154/rv-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-rv-a-rv_1784073719_2926154/rv-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-rv-a-rv_1784073719_2926154/rv-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-rv-a-rv_1784073719_2926154/rv-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-up-a-up-1784106749-3678887-up-dataset",
      "profileId": null,
      "name": "tbench-up-a-up_1784106749_3678887/up-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-up-a-up_1784106749_3678887/up-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-up-a-up_1784106749_3678887/up-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-up-a-up_1784106749_3678887/up-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-up-a-up_1784106749_3678887/up-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-up2-a-up2-1784107856-4005736-up2-dataset",
      "profileId": null,
      "name": "tbench-up2-a-up2_1784107856_4005736/up2-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-up2-a-up2_1784107856_4005736/up2-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-up2-a-up2_1784107856_4005736/up2-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-up2-a-up2_1784107856_4005736/up2-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-up2-a-up2_1784107856_4005736/up2-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-up3-a-up3-1784108188-141385-up3-dataset",
      "profileId": null,
      "name": "tbench-up3-a-up3_1784108188_141385/up3-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-up3-a-up3_1784108188_141385/up3-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-up3-a-up3_1784108188_141385/up3-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-up3-a-up3_1784108188_141385/up3-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-up3-a-up3_1784108188_141385/up3-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-tbench-up4-a-up4-1784108341-146593-up4-dataset",
      "profileId": null,
      "name": "tbench-up4-a-up4_1784108341_146593/up4-dataset",
      "category": "Harbor catalog",
      "description": "Demo/test/development artifact",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "The public name/card indicates a demo, acceptance artifact or development run; not selected for a general public evaluation suite.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/tbench-up4-a-up4_1784108341_146593/up4-dataset",
      "catalogUrl": "https://hub.harborframework.com/datasets/tbench-up4-a-up4_1784108341_146593/up4-dataset",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "tbench-up4-a-up4_1784108341_146593/up4-dataset",
          "url": "https://hub.harborframework.com/datasets/tbench-up4-a-up4_1784108341_146593/up4-dataset",
          "tasks": 1,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "harbor-withmartian-clojure-swerebench-v2-validated-51",
      "profileId": null,
      "name": "withmartian/clojure-swerebench-v2-validated-51",
      "category": "Harbor catalog",
      "description": "Specialist/watchlist; further review required",
      "metrics": "Consult the upstream task verifier.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Not promoted into the prominent shortlist from catalog visibility alone; inspect owner, task validity, evaluation and adoption first.",
      "priority": "P3",
      "maturity": "Name/count only; no substantive card text exposed",
      "sourceUrl": "https://hub.harborframework.com/datasets/withmartian/clojure-swerebench-v2-validated-51",
      "catalogUrl": "https://hub.harborframework.com/datasets/withmartian/clojure-swerebench-v2-validated-51",
      "companies": [],
      "sources": [
        {
          "kind": "Harbor",
          "name": "withmartian/clojure-swerebench-v2-validated-51",
          "url": "https://hub.harborframework.com/datasets/withmartian/clojure-swerebench-v2-validated-51",
          "tasks": 51,
          "access": "Public",
          "review": "Name/count only; no substantive card text exposed"
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-idavidrein-gpqa",
      "profileId": null,
      "name": "Idavidrein/gpqa",
      "category": "Hugging Face catalog",
      "description": "Selected science-reasoning diagnostic; Diamond is a specific subset.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/Idavidrein/gpqa",
      "catalogUrl": "https://huggingface.co/datasets/Idavidrein/gpqa",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "Idavidrein/gpqa",
          "url": "https://huggingface.co/datasets/Idavidrein/gpqa",
          "tasks": null,
          "access": "Gated",
          "review": "Selected science-reasoning diagnostic; Diamond is a specific subset."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-swe-bench-swe-bench-verified",
      "profileId": null,
      "name": "SWE-bench/SWE-bench_Verified",
      "category": "Hugging Face catalog",
      "description": "Selected primary coding-agent benchmark.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Verified",
      "catalogUrl": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Verified",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "SWE-bench/SWE-bench_Verified",
          "url": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Verified",
          "tasks": null,
          "access": "Public",
          "review": "Selected primary coding-agent benchmark."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-matharena-aime-2026",
      "profileId": null,
      "name": "MathArena/aime_2026",
      "category": "Hugging Face catalog",
      "description": "Selected recent math diagnostic; noncommercial license tag needs attention for a paid offering.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/MathArena/aime_2026",
      "catalogUrl": "https://huggingface.co/datasets/MathArena/aime_2026",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "MathArena/aime_2026",
          "url": "https://huggingface.co/datasets/MathArena/aime_2026",
          "tasks": null,
          "access": "Public",
          "review": "Selected recent math diagnostic; noncommercial license tag needs attention for a paid offering."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-tiiuae-pbench",
      "profileId": null,
      "name": "tiiuae/PBench",
      "category": "Hugging Face catalog",
      "description": "Specialist referring-expression segmentation; relevant to perception/editing products, not general agents.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/tiiuae/PBench",
      "catalogUrl": "https://huggingface.co/datasets/tiiuae/PBench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "tiiuae/PBench",
          "url": "https://huggingface.co/datasets/tiiuae/PBench",
          "tasks": null,
          "access": "Public",
          "review": "Specialist referring-expression segmentation; relevant to perception/editing products, not general agents."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-intelligencelab-long-horizon-terminal-bench",
      "profileId": null,
      "name": "IntelligenceLab/Long-Horizon-Terminal-Bench",
      "category": "Hugging Face catalog",
      "description": "Selected long-horizon terminal specialist.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/IntelligenceLab/Long-Horizon-Terminal-Bench",
      "catalogUrl": "https://huggingface.co/datasets/IntelligenceLab/Long-Horizon-Terminal-Bench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "IntelligenceLab/Long-Horizon-Terminal-Bench",
          "url": "https://huggingface.co/datasets/IntelligenceLab/Long-Horizon-Terminal-Bench",
          "tasks": null,
          "access": "Public",
          "review": "Selected long-horizon terminal specialist."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-harborframework-terminal-bench-2-1",
      "profileId": null,
      "name": "harborframework/terminal-bench-2.1",
      "category": "Hugging Face catalog",
      "description": "Selected historical-comparison release; primary source is upstream.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-2.1",
      "catalogUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-2.1",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "harborframework/terminal-bench-2.1",
          "url": "https://huggingface.co/datasets/harborframework/terminal-bench-2.1",
          "tasks": null,
          "access": "Public",
          "review": "Selected historical-comparison release; primary source is upstream."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-harborframework-terminal-bench-science",
      "profileId": null,
      "name": "harborframework/terminal-bench-science",
      "category": "Hugging Face catalog",
      "description": "Selected scientific terminal specialist.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-science",
      "catalogUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-science",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "harborframework/terminal-bench-science",
          "url": "https://huggingface.co/datasets/harborframework/terminal-bench-science",
          "tasks": null,
          "access": "Public",
          "review": "Selected scientific terminal specialist."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-lexam-benchmark-lexam",
      "profileId": null,
      "name": "LEXam-Benchmark/LEXam",
      "category": "Hugging Face catalog",
      "description": "Selected legal-exam specialist; jurisdiction matters.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/LEXam-Benchmark/LEXam",
      "catalogUrl": "https://huggingface.co/datasets/LEXam-Benchmark/LEXam",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "LEXam-Benchmark/LEXam",
          "url": "https://huggingface.co/datasets/LEXam-Benchmark/LEXam",
          "tasks": null,
          "access": "Public",
          "review": "Selected legal-exam specialist; jurisdiction matters."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-mercor-ace",
      "profileId": null,
      "name": "mercor/ACE",
      "category": "Hugging Face catalog",
      "description": "Watchlist: evaluation criteria across DIY, food, shopping and gaming; not the same as APEX-Agents.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/mercor/ACE",
      "catalogUrl": "https://huggingface.co/datasets/mercor/ACE",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "mercor/ACE",
          "url": "https://huggingface.co/datasets/mercor/ACE",
          "tasks": null,
          "access": "Public",
          "review": "Watchlist: evaluation criteria across DIY, food, shopping and gaming; not the same as APEX-Agents."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-mercor-apex-v1-extended",
      "profileId": null,
      "name": "mercor/APEX-v1-extended",
      "category": "Hugging Face catalog",
      "description": "Related professional-output benchmark; not the same task format/split as APEX-Agents 1.1.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/mercor/APEX-v1-extended",
      "catalogUrl": "https://huggingface.co/datasets/mercor/APEX-v1-extended",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "mercor/APEX-v1-extended",
          "url": "https://huggingface.co/datasets/mercor/APEX-v1-extended",
          "tasks": null,
          "access": "Public",
          "review": "Related professional-output benchmark; not the same task format/split as APEX-Agents 1.1."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-vlabench-vlabench-primitive-ft-lerobot-video",
      "profileId": null,
      "name": "VLABench/vlabench_primitive_ft_lerobot_video",
      "category": "Hugging Face catalog",
      "description": "Demonstration/fine-tuning export in LeRobot format; select the actual simulator evaluation protocol separately.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/VLABench/vlabench_primitive_ft_lerobot_video",
      "catalogUrl": "https://huggingface.co/datasets/VLABench/vlabench_primitive_ft_lerobot_video",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "VLABench/vlabench_primitive_ft_lerobot_video",
          "url": "https://huggingface.co/datasets/VLABench/vlabench_primitive_ft_lerobot_video",
          "tasks": null,
          "access": "Public",
          "review": "Demonstration/fine-tuning export in LeRobot format; select the actual simulator evaluation protocol separately."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-paddlepaddle-real5-omnidocbench",
      "profileId": null,
      "name": "PaddlePaddle/Real5-OmniDocBench",
      "category": "Hugging Face catalog",
      "description": "Selected photographed/scanned document robustness extension.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench",
      "catalogUrl": "https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "PaddlePaddle/Real5-OmniDocBench",
          "url": "https://huggingface.co/datasets/PaddlePaddle/Real5-OmniDocBench",
          "tasks": null,
          "access": "Public",
          "review": "Selected photographed/scanned document robustness extension."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-futurema-evasionbench",
      "profileId": null,
      "name": "FutureMa/EvasionBench",
      "category": "Hugging Face catalog",
      "description": "Watchlist: earnings-call evasion classification with model-generated labels; validate labels independently.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/FutureMa/EvasionBench",
      "catalogUrl": "https://huggingface.co/datasets/FutureMa/EvasionBench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "FutureMa/EvasionBench",
          "url": "https://huggingface.co/datasets/FutureMa/EvasionBench",
          "tasks": null,
          "access": "Public",
          "review": "Watchlist: earnings-call evasion classification with model-generated labels; validate labels independently."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-harborframework-terminal-bench-2-0",
      "profileId": null,
      "name": "harborframework/terminal-bench-2.0",
      "category": "Hugging Face catalog",
      "description": "Historical release; retained separately from 2.1/3.0/4.0.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-2.0",
      "catalogUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-2.0",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "harborframework/terminal-bench-2.0",
          "url": "https://huggingface.co/datasets/harborframework/terminal-bench-2.0",
          "tasks": null,
          "access": "Public",
          "review": "Historical release; retained separately from 2.1/3.0/4.0."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-collinear-ai-yc-bench",
      "profileId": null,
      "name": "collinear-ai/yc-bench",
      "category": "Hugging Face catalog",
      "description": "Watchlist: startup-management simulation; interesting long-horizon test, limited established adoption evidence.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/collinear-ai/yc-bench",
      "catalogUrl": "https://huggingface.co/datasets/collinear-ai/yc-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "collinear-ai/yc-bench",
          "url": "https://huggingface.co/datasets/collinear-ai/yc-bench",
          "tasks": null,
          "access": "Public",
          "review": "Watchlist: startup-management simulation; interesting long-horizon test, limited established adoption evidence."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-internlm-wildclawbench",
      "profileId": null,
      "name": "internlm/WildClawBench",
      "category": "Hugging Face catalog",
      "description": "Watchlist: practical OpenClaw agent tasks; assess live-service dependencies before adoption.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/internlm/WildClawBench",
      "catalogUrl": "https://huggingface.co/datasets/internlm/WildClawBench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "internlm/WildClawBench",
          "url": "https://huggingface.co/datasets/internlm/WildClawBench",
          "tasks": null,
          "access": "Public",
          "review": "Watchlist: practical OpenClaw agent tasks; assess live-service dependencies before adoption."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-claw-eval-claw-eval",
      "profileId": null,
      "name": "claw-eval/Claw-Eval",
      "category": "Hugging Face catalog",
      "description": "Watchlist: broad agent task collection; validate split-specific harness and task fidelity.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/claw-eval/Claw-Eval",
      "catalogUrl": "https://huggingface.co/datasets/claw-eval/Claw-Eval",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "claw-eval/Claw-Eval",
          "url": "https://huggingface.co/datasets/claw-eval/Claw-Eval",
          "tasks": null,
          "access": "Public",
          "review": "Watchlist: broad agent task collection; validate split-specific harness and task fidelity."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-internscience-researchclawbench",
      "profileId": null,
      "name": "InternScience/ResearchClawBench",
      "category": "Hugging Face catalog",
      "description": "Watchlist: scientific research agents; compare with established executable science suites.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/InternScience/ResearchClawBench",
      "catalogUrl": "https://huggingface.co/datasets/InternScience/ResearchClawBench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "InternScience/ResearchClawBench",
          "url": "https://huggingface.co/datasets/InternScience/ResearchClawBench",
          "tasks": null,
          "access": "Public",
          "review": "Watchlist: scientific research agents; compare with established executable science suites."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-chrishayduk-nanofold-public",
      "profileId": null,
      "name": "ChrisHayduk/nanofold-public",
      "category": "Hugging Face catalog",
      "description": "Public protein-folding train/validation data, not the private final benchmark; relevant only to protein-model researchers.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/ChrisHayduk/nanofold-public",
      "catalogUrl": "https://huggingface.co/datasets/ChrisHayduk/nanofold-public",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "ChrisHayduk/nanofold-public",
          "url": "https://huggingface.co/datasets/ChrisHayduk/nanofold-public",
          "tasks": null,
          "access": "Public",
          "review": "Public protein-folding train/validation data, not the private final benchmark; relevant only to protein-model researchers."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-benchflow-skillsbench",
      "profileId": null,
      "name": "benchflow/skillsbench",
      "category": "Hugging Face catalog",
      "description": "Selected skills evaluation; use audited versioned result protocol.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/benchflow/skillsbench",
      "catalogUrl": "https://huggingface.co/datasets/benchflow/skillsbench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "benchflow/skillsbench",
          "url": "https://huggingface.co/datasets/benchflow/skillsbench",
          "tasks": null,
          "access": "Public",
          "review": "Selected skills evaluation; use audited versioned result protocol."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-harborframework-terminal-bench-3-0",
      "profileId": null,
      "name": "harborframework/terminal-bench-3.0",
      "category": "Hugging Face catalog",
      "description": "Intermediate release; keep separately versioned; current primary recommendation also includes 4.0.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-3.0",
      "catalogUrl": "https://huggingface.co/datasets/harborframework/terminal-bench-3.0",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "harborframework/terminal-bench-3.0",
          "url": "https://huggingface.co/datasets/harborframework/terminal-bench-3.0",
          "tasks": null,
          "access": "Public",
          "review": "Intermediate release; keep separately versioned; current primary recommendation also includes 4.0."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "hf-harborframework-terminal-bench",
      "profileId": null,
      "name": "harborframework/terminal-bench",
      "category": "Hugging Face catalog",
      "description": "Selected continuous benchmark family; pin an immutable release rather than main.",
      "metrics": "Consult the upstream evaluation protocol.",
      "execution": "Dataset-specific harness qualification required.",
      "limitations": "Badge is catalog metadata, not a maturity, licensing or runnable-harness certification.",
      "priority": "P3",
      "maturity": "Official-filter dataset card",
      "sourceUrl": "https://huggingface.co/datasets/harborframework/terminal-bench",
      "catalogUrl": "https://huggingface.co/datasets/harborframework/terminal-bench",
      "companies": [],
      "sources": [
        {
          "kind": "Hugging Face",
          "name": "harborframework/terminal-bench",
          "url": "https://huggingface.co/datasets/harborframework/terminal-bench",
          "tasks": null,
          "access": "Public",
          "review": "Selected continuous benchmark family; pin an immutable release rather than main."
        }
      ],
      "hostedId": null,
      "productHref": null,
      "productResultState": null
    },
    {
      "id": "arc-crm-6",
      "profileId": null,
      "name": "Arc CRM 6",
      "category": "CRM operations",
      "description": "Independent synthetic CRM work across contact onboarding, stage correction, bounded quotes, quote replacement, unsigned contracts and corrected document associations. All source conversations were inspected; zero source rows are reproduced.",
      "metrics": "Deterministic evidence, post-write readback, outcome, containment, handoff and structured-answer checks over saved SQLite state. All 167 authoring negative controls reject. Reference controls establish solvability, not model performance.",
      "execution": "Package 0.1.2 exposes 31 domain/evidence contracts plus context and answer controls through CLI, real HTML forms, REST and MCP sharing an isolated episode-local world. A separate networkless verifier grades the collected state. Six source-inspired workflows and 33 synthetic evidence files (PDF, XLSX, EML and Markdown) are partial coverage, not 1,200 converted conversations or upstream API parity.",
      "limitations": "This Blobfish suite has its own task and environment definitions. Upstream scores and reference controls are not company performance scores.",
      "priority": "P2",
      "maturity": "Blobfish benchmark suite",
      "sourceUrl": "https://github.com/blobfishai/arc-crm-agent-simulation/tree/090d4028e485d21190588006645a13e125aef130",
      "catalogUrl": "https://hub.harborframework.com/datasets/blobfishai/arc-crm-6/v0.1.2",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": "/benchmarks/arc-crm-6",
      "productResultState": "awaiting-first-model-run"
    },
    {
      "id": "filingbench-50",
      "profileId": null,
      "name": "FilingBench-50",
      "category": "Financial research",
      "description": "Public financial-research questions paired with a deterministic DCF, assumptions, period-data, and review-note world.",
      "metrics": "Tool-assisted answers to complex company, financial-statement, and SEC-filing questions.",
      "execution": "The upstream runner may require provider, search, and SEC data credentials.",
      "limitations": "This Blobfish suite has its own task and environment definitions. Upstream scores and reference controls are not company performance scores.",
      "priority": "P2",
      "maturity": "Blobfish benchmark suite",
      "sourceUrl": "https://github.com/vals-ai/finance-agent",
      "catalogUrl": "https://hub.harborframework.com/datasets/vals/financeagent",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": "/benchmarks/filingbench-50",
      "productResultState": "baseline-evidence-live"
    },
    {
      "id": "servicebench-375",
      "profileId": null,
      "name": "ServiceBench-375",
      "category": "Customer operations",
      "description": "Multi-turn service-agent evaluation paired with an SLA, ticket, escalation, message, and knowledge-base world.",
      "metrics": "Policy-aware, tool-using conversations across airline, retail, telecom, and banking knowledge domains.",
      "execution": "The official repository remains tau2-bench; its current release is branded τ³-bench.",
      "limitations": "This Blobfish suite has its own task and environment definitions. Upstream scores and reference controls are not company performance scores.",
      "priority": "P2",
      "maturity": "Blobfish benchmark suite",
      "sourceUrl": "https://github.com/sierra-research/tau2-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/sierra-research/tau3-bench",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": "/benchmarks/servicebench-375",
      "productResultState": "baseline-evidence-live"
    },
    {
      "id": "discoverybench-102",
      "profileId": null,
      "name": "DiscoveryBench-102",
      "category": "Scientific discovery",
      "description": "Scientific-programming tasks paired with an assay, compound, target, QC-result, and literature world.",
      "metrics": "102 scientific-programming tasks: 38 deterministic and 64 visual/LLM-judge tasks.",
      "execution": "Three tasks require GPUs; visual tasks require the verifier model configured by the upstream package.",
      "limitations": "This Blobfish suite has its own task and environment definitions. Upstream scores and reference controls are not company performance scores.",
      "priority": "P2",
      "maturity": "Blobfish benchmark suite",
      "sourceUrl": "https://github.com/OSU-NLP-Group/ScienceAgentBench",
      "catalogUrl": "https://hub.harborframework.com/datasets/scienceagentbench/scienceagentbench",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": "/benchmarks/discoverybench-102",
      "productResultState": "baseline-evidence-live"
    },
    {
      "id": "defensebench-100",
      "profileId": null,
      "name": "DefenseBench-100",
      "category": "Defensive security",
      "description": "Local defensive-hardening tasks paired with a permission-aware search, document, employee, and access-grant world.",
      "metrics": "Defensive code hardening against locally simulated vulnerability probes.",
      "execution": "The upstream task package uses local sandboxes and does not require external targets.",
      "limitations": "This Blobfish suite has its own task and environment definitions. Upstream scores and reference controls are not company performance scores.",
      "priority": "P2",
      "maturity": "Blobfish benchmark suite",
      "sourceUrl": "https://hub.harborframework.com/datasets/polyvorlabs/cyberdefense-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/polyvorlabs/cyberdefense-bench",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": "/benchmarks/defensebench-100",
      "productResultState": "baseline-evidence-live"
    },
    {
      "id": "opsbench-14",
      "profileId": null,
      "name": "OpsBench-14",
      "category": "Enterprise operations",
      "description": "Cross-functional enterprise questions paired with vendors, processes, changes, controls, and review workflows.",
      "metrics": "Cross-system retrieval and analytical tasks over production-scale enterprise data.",
      "execution": "Upstream scoring uses required criteria and repeated trials; its published methodology should remain authoritative.",
      "limitations": "This Blobfish suite has its own task and environment definitions. Upstream scores and reference controls are not company performance scores.",
      "priority": "P2",
      "maturity": "Blobfish benchmark suite",
      "sourceUrl": "https://github.com/devrev/enterprise-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/Enterprise-Bench/l1-l2-bench",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": "/benchmarks/opsbench-14",
      "productResultState": "baseline-evidence-live"
    },
    {
      "id": "repobench-500",
      "profileId": null,
      "name": "RepoBench-500",
      "category": "Software engineering",
      "description": "Expert-verified GitHub issues paired with a service, repository, pull-request, incident, ADR, and standup world.",
      "metrics": "Expert-verified, solvable real-world GitHub issues scored by repository tests.",
      "execution": "The upstream task environments use repository-specific Docker images and test contracts.",
      "limitations": "This Blobfish suite has its own task and environment definitions. Upstream scores and reference controls are not company performance scores.",
      "priority": "P2",
      "maturity": "Blobfish benchmark suite",
      "sourceUrl": "https://github.com/SWE-bench/SWE-bench",
      "catalogUrl": "https://hub.harborframework.com/datasets/swe-bench/swe-bench-verified",
      "companies": [],
      "sources": [],
      "hostedId": null,
      "productHref": "/benchmarks/repobench-500",
      "productResultState": "baseline-evidence-live"
    }
  ]
}
