From 7b8000cbe9cf7b48668f93c2b0b822ba91f9588c Mon Sep 17 00:00:00 2001 From: JBAhire Date: Wed, 10 Jun 2026 10:13:55 -0700 Subject: [PATCH] =?UTF-8?q?feat:=20June=202026=20data=20refresh=20?= =?UTF-8?q?=E2=80=94=2050=20new=20evaluations,=2045+=20status=20correction?= =?UTF-8?q?s?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Models (22 new): Claude Fable 5, Opus 4.8/4.7/4.6, Sonnet 4.6, GPT-5.5, GPT-5.4, GPT-5.3-Codex, Gemini 3.1 Pro, Gemini 3.5 Flash, Grok 4.3/4.1, Nova 2 Lite, DeepSeek V3.2/V4, Qwen3.5, Kimi K2.6, GLM-5, MiniMax-M2, Mistral Large 3, Gemma 4, Command A+. Agents (16 new): Claude Code, Claude Agent SDK, OpenAI Agents SDK, Codex, Google ADK, Gemini CLI, Jules, GitHub Copilot coding agent, Microsoft Agent Framework, Devin, Cursor, Manus, smolagents, Strands, Mastra, Dify. MCPs (12 new): Context7, Chrome DevTools, Playwright, Serena, Figma, Stripe, Vercel, Hugging Face, Firecrawl, shadcn, Apify, Zapier. Status corrections: Grok 3 retired, Gemini 2.0 Flash shut down, Assistants API sunset 2026-08-26, OpenAI o-series/GPT-5.x deprecation dates, archived MCP reference servers flagged (incl. unpatched SQLi in postgres/sqlite), Langflow CVEs, MemGPT->Letta, Agent Builder rebrand, AutoGen/SK maintenance mode, Llama 4 Behemoth marked never released. --- README.md | 69 +- data/agents/agentgpt.json | 40 +- data/agents/amazon-bedrock-agents.json | 18 +- data/agents/autogen.json | 21 +- data/agents/babyagi.json | 23 +- data/agents/claude-agent-sdk.json | 465 ++++++++++++ data/agents/claude-code.json | 479 +++++++++++++ data/agents/cursor-agent.json | 463 ++++++++++++ data/agents/devin.json | 462 ++++++++++++ data/agents/dify.json | 479 +++++++++++++ data/agents/gemini-cli.json | 458 ++++++++++++ data/agents/github-copilot-coding-agent.json | 455 ++++++++++++ data/agents/google-adk.json | 471 +++++++++++++ data/agents/google-agent-builder.json | 24 +- data/agents/google-jules.json | 454 ++++++++++++ data/agents/langflow.json | 51 +- data/agents/langgraph-agent.json | 22 +- data/agents/manus.json | 466 ++++++++++++ data/agents/mastra.json | 473 +++++++++++++ data/agents/memgpt.json | 24 +- data/agents/microsoft-agent-framework.json | 469 ++++++++++++ data/agents/openai-agents-sdk.json | 467 ++++++++++++ data/agents/openai-assistants-api.json | 30 +- data/agents/openai-codex.json | 479 +++++++++++++ data/agents/pydantic-ai.json | 20 +- data/agents/salesforce-einstein-bots.json | 18 +- data/agents/semantic-kernel-agent.json | 21 +- data/agents/smolagents.json | 475 +++++++++++++ data/agents/strands-agents.json | 477 +++++++++++++ data/agents/swarm.json | 26 +- data/mcps/mcp-server-apify.json | 450 ++++++++++++ data/mcps/mcp-server-brave-search.json | 42 +- data/mcps/mcp-server-chrome-devtools.json | 443 ++++++++++++ data/mcps/mcp-server-context7.json | 464 ++++++++++++ data/mcps/mcp-server-everything.json | 23 +- data/mcps/mcp-server-fetch.json | 23 +- data/mcps/mcp-server-figma.json | 436 ++++++++++++ data/mcps/mcp-server-filesystem.json | 23 +- data/mcps/mcp-server-firecrawl.json | 455 ++++++++++++ data/mcps/mcp-server-git.json | 23 +- data/mcps/mcp-server-github.json | 164 +++-- data/mcps/mcp-server-gitlab.json | 43 +- data/mcps/mcp-server-google-drive.json | 47 +- data/mcps/mcp-server-hugging-face.json | 436 ++++++++++++ data/mcps/mcp-server-memory.json | 21 +- data/mcps/mcp-server-notion.json | 109 +-- data/mcps/mcp-server-playwright.json | 452 ++++++++++++ data/mcps/mcp-server-postgres.json | 95 ++- data/mcps/mcp-server-puppeteer.json | 49 +- data/mcps/mcp-server-sequential-thinking.json | 23 +- data/mcps/mcp-server-serena.json | 444 ++++++++++++ data/mcps/mcp-server-shadcn.json | 421 +++++++++++ data/mcps/mcp-server-slack.json | 45 +- data/mcps/mcp-server-sqlite.json | 58 +- data/mcps/mcp-server-stripe.json | 440 ++++++++++++ data/mcps/mcp-server-time.json | 21 +- data/mcps/mcp-server-vercel.json | 433 ++++++++++++ data/mcps/mcp-server-zapier.json | 444 ++++++++++++ data/models/claude-fable-5.json | 651 +++++++++++++++++ data/models/claude-opus-4-5.json | 19 +- data/models/claude-opus-4-6.json | 661 +++++++++++++++++ data/models/claude-opus-4-7.json | 650 +++++++++++++++++ data/models/claude-opus-4-8.json | 652 +++++++++++++++++ data/models/claude-sonnet-4-6.json | 647 +++++++++++++++++ data/models/command-a-plus.json | 639 +++++++++++++++++ data/models/deepseek-r1.json | 27 +- data/models/deepseek-v3-0324.json | 27 +- data/models/deepseek-v3-2.json | 638 +++++++++++++++++ data/models/deepseek-v4.json | 632 +++++++++++++++++ data/models/gemini-2-0-flash.json | 23 +- data/models/gemini-3-1-pro.json | 627 +++++++++++++++++ data/models/gemini-3-5-flash.json | 608 ++++++++++++++++ data/models/gemini-3-pro.json | 20 +- data/models/gemma-3-27b.json | 22 +- data/models/gemma-4.json | 595 ++++++++++++++++ data/models/glm-5.json | 644 +++++++++++++++++ data/models/gpt-4o.json | 23 +- data/models/gpt-5-1.json | 16 +- data/models/gpt-5-2-codex.json | 24 +- data/models/gpt-5-2.json | 19 +- data/models/gpt-5-3-codex.json | 605 ++++++++++++++++ data/models/gpt-5-4.json | 644 +++++++++++++++++ data/models/gpt-5-5.json | 665 ++++++++++++++++++ data/models/grok-3-beta.json | 30 +- data/models/grok-4-1.json | 633 +++++++++++++++++ data/models/grok-4-3.json | 636 +++++++++++++++++ data/models/kimi-k2-6.json | 644 +++++++++++++++++ data/models/llama-3-1-405b.json | 7 +- data/models/llama-3-3-70b.json | 7 +- data/models/llama-4-behemoth.json | 53 +- data/models/minimax-m2.json | 638 +++++++++++++++++ data/models/mistral-large-3.json | 632 +++++++++++++++++ data/models/nemotron-ultra-253b.json | 21 +- data/models/nova-2-lite.json | 626 +++++++++++++++++ data/models/nova-pro.json | 15 +- data/models/openai-o1-mini.json | 23 +- data/models/openai-o1.json | 24 +- data/models/openai-o3-mini.json | 24 +- data/models/openai-o3.json | 24 +- data/models/openai-o4-mini.json | 26 +- data/models/qwen2-5-vl-32b.json | 26 +- data/models/qwen3-5.json | 644 +++++++++++++++++ lib/data.ts | 150 +++- 103 files changed, 28090 insertions(+), 597 deletions(-) create mode 100644 data/agents/claude-agent-sdk.json create mode 100644 data/agents/claude-code.json create mode 100644 data/agents/cursor-agent.json create mode 100644 data/agents/devin.json create mode 100644 data/agents/dify.json create mode 100644 data/agents/gemini-cli.json create mode 100644 data/agents/github-copilot-coding-agent.json create mode 100644 data/agents/google-adk.json create mode 100644 data/agents/google-jules.json create mode 100644 data/agents/manus.json create mode 100644 data/agents/mastra.json create mode 100644 data/agents/microsoft-agent-framework.json create mode 100644 data/agents/openai-agents-sdk.json create mode 100644 data/agents/openai-codex.json create mode 100644 data/agents/smolagents.json create mode 100644 data/agents/strands-agents.json create mode 100644 data/mcps/mcp-server-apify.json create mode 100644 data/mcps/mcp-server-chrome-devtools.json create mode 100644 data/mcps/mcp-server-context7.json create mode 100644 data/mcps/mcp-server-figma.json create mode 100644 data/mcps/mcp-server-firecrawl.json create mode 100644 data/mcps/mcp-server-hugging-face.json create mode 100644 data/mcps/mcp-server-playwright.json create mode 100644 data/mcps/mcp-server-serena.json create mode 100644 data/mcps/mcp-server-shadcn.json create mode 100644 data/mcps/mcp-server-stripe.json create mode 100644 data/mcps/mcp-server-vercel.json create mode 100644 data/mcps/mcp-server-zapier.json create mode 100644 data/models/claude-fable-5.json create mode 100644 data/models/claude-opus-4-6.json create mode 100644 data/models/claude-opus-4-7.json create mode 100644 data/models/claude-opus-4-8.json create mode 100644 data/models/claude-sonnet-4-6.json create mode 100644 data/models/command-a-plus.json create mode 100644 data/models/deepseek-v3-2.json create mode 100644 data/models/deepseek-v4.json create mode 100644 data/models/gemini-3-1-pro.json create mode 100644 data/models/gemini-3-5-flash.json create mode 100644 data/models/gemma-4.json create mode 100644 data/models/glm-5.json create mode 100644 data/models/gpt-5-3-codex.json create mode 100644 data/models/gpt-5-4.json create mode 100644 data/models/gpt-5-5.json create mode 100644 data/models/grok-4-1.json create mode 100644 data/models/grok-4-3.json create mode 100644 data/models/kimi-k2-6.json create mode 100644 data/models/minimax-m2.json create mode 100644 data/models/mistral-large-3.json create mode 100644 data/models/nova-2-lite.json create mode 100644 data/models/qwen3-5.json diff --git a/README.md b/README.md index af80b93..05a962c 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) [![PRs Welcome](https://img.shields.io/badge/PRs-welcome-brightgreen.svg)](http://makeapullrequest.com) -[![Evaluations](https://img.shields.io/badge/Evaluations-106-blue.svg)](#-current-coverage) +[![Evaluations](https://img.shields.io/badge/Evaluations-156-blue.svg)](#-current-coverage) [![GitHub Stars](https://img.shields.io/github/stars/Guard0-Security/TrustVector?style=social)](https://github.com/Guard0-Security/TrustVector) TrustVector is an evidence-based evaluation framework for AI systems, providing transparent, multi-dimensional trust scores across **security**, **privacy**, **performance**, **trust**, and **operational excellence**. @@ -87,57 +87,54 @@ const customScore = calculateCustomScore(claudeSonnet, { ## 📊 Current Coverage -**106 Total Evaluations** across 3 categories: +**156 Total Evaluations** across 3 categories (last refreshed June 2026): -### AI Models (38) +### AI Models (60) **Frontier Models:** -- ✅ Claude Sonnet 4.5, Claude Opus 4.1, Claude 3.7 Sonnet, Claude 3.5 Haiku (Anthropic) -- ✅ GPT-5, GPT-4.5, GPT-4.1, GPT-4o, GPT-4o Mini (OpenAI) -- ✅ o1, o1 Mini, o3, o3 Mini (OpenAI Reasoning) -- ✅ Gemini 2.5 Pro, Gemini 2.0 Flash (Google) -- ✅ Llama 4 Behemoth, Llama 4 Maverick, Llama 4 Scout, Llama 3.3 70B, Llama 3.1 405B (Meta) -- ✅ Grok 3 Beta (xAI) -- ✅ DeepSeek R1, DeepSeek V3 (DeepSeek) - -**Specialized & Open Source:** -- ✅ Gemma 3 27B (Google) -- ✅ Qwen2.5-VL 32B (Alibaba) -- ✅ Nemotron Ultra 253B (NVIDIA) -- ✅ Nova Pro (Amazon) +- ✅ Claude Fable 5, Claude Opus 4.8 / 4.7 / 4.6 / 4.5, Claude Sonnet 4.6 / 4.5, Claude Haiku 4.5 (Anthropic) +- ✅ GPT-5.5, GPT-5.4, GPT-5.3-Codex, GPT-5.2, GPT-5.1, GPT-5, o-series (OpenAI) +- ✅ Gemini 3.1 Pro, Gemini 3.5 Flash, Gemini 3 Pro/Flash (Google) +- ✅ Grok 4.3, Grok 4.1 (xAI) +- ✅ Nova 2 Lite, Nova Pro (Amazon) + +**Open-Weight Models:** +- ✅ DeepSeek V4, DeepSeek V3.2, DeepSeek R1 (DeepSeek) +- ✅ Qwen3.5 (Alibaba), Kimi K2.6 (Moonshot), GLM-5 (Z.ai), MiniMax-M2 +- ✅ Mistral Large 3 (Mistral), Command A+ (Cohere) +- ✅ Gemma 4, Gemma 3 (Google), gpt-oss-120b/20b (OpenAI) +- ✅ Llama 4 Maverick/Scout, Llama 3.x (Meta), Nemotron (NVIDIA) **[See all models →](/data/models)** -### AI Agents (34) +### AI Agents (50) -**Enterprise Platforms:** -- ✅ Amazon Bedrock Agents, Azure Bot Service, Google Agent Builder -- ✅ IBM Watson Assistant, Google Dialogflow, Amazon Lex +**Coding & Autonomous Agents:** +- ✅ Claude Code + Claude Agent SDK (Anthropic), OpenAI Codex, Devin (Cognition) +- ✅ Cursor, GitHub Copilot coding agent, Google Jules, Gemini CLI, Manus **Developer Frameworks:** -- ✅ LangGraph Agent, LlamaIndex Agent, CrewAI, AutoGen -- ✅ Haystack, LangFlow, Flowise, E2B Agents +- ✅ OpenAI Agents SDK, Google ADK, Microsoft Agent Framework, AWS Strands Agents +- ✅ LangGraph, CrewAI, LlamaIndex, Pydantic AI, smolagents, Mastra, Dify -**Autonomous Agents:** -- ✅ AutoGPT, BabyAGI, AgentGPT, Adala -- ✅ And 15+ more... +**Enterprise Platforms:** +- ✅ Amazon Bedrock Agents, Azure Bot Service, Gemini Enterprise Agent Platform +- ✅ IBM watsonx Assistant, Google Dialogflow, Amazon Lex, and more **[See all agents →](/data/agents)** -### MCP Servers (34) - -**Cloud & Infrastructure:** -- ✅ AWS, Azure, Cloudflare, Docker, Kubernetes +### MCP Servers (46) -**Development Tools:** -- ✅ GitHub, Git, Filesystem, Memory +**Top Ecosystem Servers:** +- ✅ Context7, Chrome DevTools MCP, Playwright MCP, Serena -**Productivity & Business:** -- ✅ Gmail, Google Drive, Calendar, Linear, Atlassian -- ✅ Datadog, Elasticsearch, MongoDB +**Official Vendor Servers:** +- ✅ GitHub, Figma, Stripe, Notion, Vercel, Hugging Face, Zapier, Apify -**Utilities:** -- ✅ Brave Search, Fetch, Everything +**Reference & Community:** +- ✅ Fetch, Git, Filesystem, Memory, Sequential Thinking, Time, Everything +- ✅ AWS, Azure, Cloudflare, Docker, Kubernetes, databases, and more +- ⚠️ Archived reference servers (Puppeteer, Postgres, SQLite, Slack, …) are flagged with security advisories - ✅ And 15+ more... **[See all MCPs →](/data/mcps)** diff --git a/data/agents/agentgpt.json b/data/agents/agentgpt.json index 588ed11..2b66ab4 100644 --- a/data/agents/agentgpt.json +++ b/data/agents/agentgpt.json @@ -4,9 +4,9 @@ "name": "AgentGPT", "provider": "Reworkd", "version": "Platform", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Browser-based autonomous AI agent platform for deploying and managing GPT-powered agents. Enables users to create goal-oriented autonomous agents that break down objectives and execute tasks without continuous human intervention.", + "description": "DISCONTINUED: the AgentGPT repository was archived on 2026-01-28 (last release v1.0.0, Nov 2023) and the hosted site is frozen. Formerly a browser-based autonomous AI agent platform that let users create goal-oriented agents which break down objectives and execute tasks without continuous human intervention. Not recommended for new use.", "website": "https://agentgpt.reworkd.ai/", "trust_vector": { "performance_reliability": { @@ -249,7 +249,7 @@ } }, "trust_transparency": { - "overall_score": 79, + "overall_score": 76, "criteria": { "documentation_quality": { "score": 75, @@ -308,23 +308,30 @@ "last_verified": "2025-11-09" }, "community_support": { - "score": 73, - "confidence": "medium", + "score": 55, + "confidence": "high", "evidence": [ { "source": "Community", "url": "https://github.com/reworkd/AgentGPT/discussions", "date": "2024-10-15", "value": "Active GitHub community and Discord" + }, + { + "source": "GitHub Repository Status", + "url": "https://github.com/reworkd/AgentGPT", + "date": "2026-06-10", + "value": "Repository archived 2026-01-28; community activity has ceased" } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced: repository archived, no further community development" } } }, "operational_excellence": { - "overall_score": 70, + "overall_score": 65, "criteria": { "ease_of_use": { "score": 88, @@ -383,18 +390,25 @@ "last_verified": "2025-11-09" }, "production_readiness": { - "score": 60, - "confidence": "medium", + "score": 40, + "confidence": "high", "evidence": [ { "source": "Platform Maturity", "url": "https://github.com/reworkd/AgentGPT", "date": "2024-10-01", "value": "Experimental platform, not designed for production use" + }, + { + "source": "GitHub Repository Status", + "url": "https://github.com/reworkd/AgentGPT", + "date": "2026-06-10", + "value": "Repository archived 2026-01-28; last release v1.0.0 (Nov 2023); hosted site frozen" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced: project discontinued and repository archived on 2026-01-28" }, "reliability": { "score": 62, @@ -474,7 +488,8 @@ "Limited tool access and action capabilities", "Can be expensive with unpredictable OpenAI API costs", "No enterprise features (auth, monitoring, management)", - "Reliability and success rate varies significantly" + "Reliability and success rate varies significantly", + "Discontinued: repository archived 2026-01-28, hosted site frozen, no further updates" ], "metadata": { "license": "GPL-3.0", @@ -500,6 +515,7 @@ }, "tags": [ "autonomous", - "web-based" + "web-based", + "archived" ] } diff --git a/data/agents/amazon-bedrock-agents.json b/data/agents/amazon-bedrock-agents.json index 0fe273e..d6bdead 100644 --- a/data/agents/amazon-bedrock-agents.json +++ b/data/agents/amazon-bedrock-agents.json @@ -4,9 +4,9 @@ "name": "Amazon Bedrock Agents", "provider": "Amazon Web Services", "version": "2024", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Fully managed AWS service for building and deploying generative AI agents. Handles orchestration, memory, knowledge bases, and action groups with enterprise security and scalability built-in.", + "description": "Fully managed AWS service for building and deploying generative AI agents with orchestration, memory, knowledge bases, and action groups. Note: AWS's strategic agent runtime is now Bedrock AgentCore (GA 2025-10-13; framework-agnostic, 8-hour sessions, session isolation) paired with the open-source Strands Agents SDK; evaluate AgentCore for new builds.", "website": "https://aws.amazon.com/bedrock/agents/", "trust_vector": { "performance_reliability": { @@ -378,10 +378,16 @@ "url": "https://aws.amazon.com/bedrock/customers/", "date": "2024-10-01", "value": "Production-ready managed service with enterprise customers" + }, + { + "source": "Amazon Bedrock AgentCore GA Announcement", + "url": "https://aws.amazon.com/about-aws/whats-new/2025/10/amazon-bedrock-agentcore-available/", + "date": "2026-06-10", + "value": "AWS's strategic agent runtime is now Bedrock AgentCore (GA 2025-10-13): framework-agnostic, 8-hour sessions, session isolation; complemented by the open-source Strands Agents SDK" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -448,7 +454,8 @@ "Limited to AWS Bedrock foundation models", "Less flexibility than code-based frameworks", "Requires AWS expertise for optimal configuration", - "Not open source, limited customization of core orchestration" + "Not open source, limited customization of core orchestration", + "AWS's strategic focus has shifted to Bedrock AgentCore (GA 2025-10-13) and the Strands Agents SDK; new agent workloads should evaluate AgentCore first" ], "metadata": { "license": "Proprietary (AWS)", @@ -471,6 +478,9 @@ "regions_available": "Multiple AWS regions", "sla": "99.9%" }, + "related_entities": [ + "strands-agents" + ], "tags": [ "aws" ] diff --git a/data/agents/autogen.json b/data/agents/autogen.json index 49a07b5..d894548 100644 --- a/data/agents/autogen.json +++ b/data/agents/autogen.json @@ -4,9 +4,9 @@ "name": "Microsoft AutoGen", "provider": "Microsoft Research", "version": "0.4", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Multi-agent conversation framework enabling next-gen LLM applications with conversable agents that can operate in various modes combining LLMs, human inputs, and tools. Supports complex workflows through agent conversations.", + "description": "MAINTENANCE MODE: AutoGen now receives bug/security fixes only and is superseded by the Microsoft Agent Framework (1.0 GA on 2026-04-03), the recommended migration path. AutoGen is a multi-agent conversation framework for LLM applications with conversable agents combining LLMs, human input, and tools across complex workflows.", "website": "https://microsoft.github.io/autogen/", "trust_vector": { "performance_reliability": { @@ -377,10 +377,16 @@ "url": "https://github.com/microsoft/autogen", "date": "2024-10-20", "value": "Very active community with Microsoft backing" + }, + { + "source": "Microsoft Agent Framework Migration Guidance", + "url": "https://devblogs.microsoft.com/semantic-kernel/migrate-your-semantic-kernel-and-autogen-projects-to-microsoft-agent-framework-release-candidate/", + "date": "2026-06-10", + "value": "AutoGen is in maintenance mode (bug/security fixes only); Microsoft Agent Framework 1.0 reached GA on 2026-04-03 as the successor" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -447,7 +453,8 @@ "Requires careful prompt engineering for agent roles", "Limited built-in persistence for long-running workflows", "Some learning curve for advanced features", - "Performance depends heavily on LLM quality" + "Performance depends heavily on LLM quality", + "Maintenance mode: bug/security fixes only; new development targets Microsoft Agent Framework" ], "metadata": { "license": "Apache 2.0", @@ -474,9 +481,13 @@ "contributors": "559+", "transition_notice": "Microsoft Agent Framework is the recommended path forward; AutoGen receives maintenance and critical patches only" }, + "related_entities": [ + "microsoft-agent-framework" + ], "tags": [ "multi-agent", "microsoft", - "open-source" + "open-source", + "maintenance-mode" ] } diff --git a/data/agents/babyagi.json b/data/agents/babyagi.json index 5cdf570..ff15db2 100644 --- a/data/agents/babyagi.json +++ b/data/agents/babyagi.json @@ -4,9 +4,9 @@ "name": "BabyAGI", "provider": "Yohei Nakajima", "version": "Classic", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Minimalist autonomous task-driven AI agent that creates, prioritizes, and executes tasks based on results of previous tasks and a predefined objective. Demonstrates AGI concepts in under 200 lines of code.", + "description": "ARCHIVED: the original BabyAGI repo was archived to babyagi_archive in September 2024 and replaced by an experimental self-building framework; it is not production-maintained. Originally a minimalist autonomous task-driven AI agent that created, prioritized, and executed tasks toward an objective, demonstrating AGI concepts in under 200 lines of code.", "website": "https://github.com/yoheinakajima/babyagi", "trust_vector": { "performance_reliability": { @@ -310,7 +310,7 @@ } }, "operational_excellence": { - "overall_score": 61, + "overall_score": 57, "criteria": { "ease_of_integration": { "score": 75, @@ -370,7 +370,7 @@ "last_verified": "2025-11-09" }, "production_readiness": { - "score": 50, + "score": 35, "confidence": "high", "evidence": [ { @@ -378,10 +378,17 @@ "url": "https://github.com/yoheinakajima/babyagi", "date": "2024-09-01", "value": "Designed as concept demonstration, not production system" + }, + { + "source": "GitHub Repository Status", + "url": "https://github.com/yoheinakajima/babyagi", + "date": "2026-06-10", + "value": "Original repo archived to babyagi_archive (Sept 2024); replaced by an experimental self-building framework; not production-maintained" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced: original project archived and unmaintained since September 2024" } } } @@ -400,7 +407,8 @@ "Can generate excessive tasks leading to high costs", "No built-in security or sandboxing features", "Limited tool integration in classic version", - "Unpredictable behavior and task completion quality" + "Unpredictable behavior and task completion quality", + "Archived (Sept 2024): original repo moved to babyagi_archive with no further maintenance" ], "metadata": { "license": "MIT", @@ -471,6 +479,7 @@ "tags": [ "autonomous", "experimental", - "open-source" + "open-source", + "archived" ] } diff --git a/data/agents/claude-agent-sdk.json b/data/agents/claude-agent-sdk.json new file mode 100644 index 0000000..0665a57 --- /dev/null +++ b/data/agents/claude-agent-sdk.json @@ -0,0 +1,465 @@ +{ + "id": "claude-agent-sdk", + "type": "agent", + "name": "Claude Agent SDK", + "provider": "Anthropic", + "version": "0.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "SDK exposing Claude Code's production agent harness (tool loop, permission system, subagents, MCP) for building general-purpose agents in TypeScript and Python. Renamed from Claude Code SDK in September 2025.", + "website": "https://code.claude.com/docs/en/agent-sdk/overview", + "trust_vector": { + "performance_reliability": { + "overall_score": 86, + "criteria": { + "task_completion_accuracy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK overview", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Reuses the same battle-tested agent harness that powers Claude Code, inheriting its task-completion behavior" + } + ], + "methodology": "Evaluation of agents built on the SDK harness across coding and non-coding tasks", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Built-in file, bash, and web tools plus custom tools and MCP servers with a managed agentic tool loop" + } + ], + "methodology": "Testing of built-in tool loop, custom tool definitions, and MCP server integration", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Automatic context management and compaction enable long-horizon multi-step agent runs" + } + ], + "methodology": "Long-horizon agent task evaluation using the SDK's managed loop and compaction", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK sessions documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/sessions", + "date": "2026-04-20", + "value": "Session persistence and resume, CLAUDE.md-style memory files, and automatic compaction of long contexts" + } + ], + "methodology": "Review of session resume, memory file support, and context compaction behavior", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Harness retries and self-corrects on tool failures; hooks allow custom failure handling" + } + ], + "methodology": "Observed recovery from tool errors and failed commands in SDK-built agents", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK subagents documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/subagents", + "date": "2026-04-20", + "value": "First-class subagents with isolated contexts, custom prompts, and per-agent tool restrictions" + } + ], + "methodology": "Testing of programmatic subagent definition and parallel delegation", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 78, + "criteria": { + "tool_sandboxing": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK permissions documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/sdk-permissions", + "date": "2026-04-20", + "value": "Inherits Claude Code's sandboxing support, including OS-level filesystem and network isolation for bash" + } + ], + "methodology": "Review of inherited sandbox configuration and bash isolation options", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK permissions documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/sdk-permissions", + "date": "2026-04-20", + "value": "Programmatic permission modes, allow/deny rules, canUseTool callbacks, and hooks for fine-grained control" + } + ], + "methodology": "Assessment of permission rules, programmatic approval callbacks, and hook-based gating", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Permission gating and sandboxing mitigate injection impact, but defenses against untrusted content depend on developer configuration" + } + ], + "methodology": "Review of documented mitigations and developer responsibility for untrusted input handling", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK sessions documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/sessions", + "date": "2026-04-20", + "value": "Sessions and subagents maintain isolated contexts; deployment isolation is the integrator's responsibility" + } + ], + "methodology": "Architecture review of session/subagent context isolation in self-hosted deployments", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "@anthropic-ai/claude-agent-sdk on npm", + "url": "https://www.npmjs.com/package/@anthropic-ai/claude-agent-sdk", + "date": "2026-06-01", + "value": "Distributed under Anthropic Commercial Terms (proprietary); SDK wrapper code inspectable but the underlying harness is closed" + } + ], + "methodology": "License and source availability review of npm/PyPI packages", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 74, + "criteria": { + "data_retention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic privacy policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-03-01", + "value": "API traffic not used for training by default; agent state and transcripts stay on the integrator's infrastructure" + } + ], + "methodology": "Review of Anthropic API retention terms and local-state architecture", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-03-01", + "value": "SOC 2 Type II and GDPR-aligned DPA available; Bedrock/Vertex routing supports regional data residency" + } + ], + "methodology": "Compliance certification review including cloud-provider routing options", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Data flows only to the configured Claude endpoint (Anthropic API, Bedrock, or Vertex AI)" + } + ], + "methodology": "Data flow analysis across supported model endpoints", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Harness runs entirely on your infrastructure, but inference requires Claude via API, Bedrock, or Vertex; no local models" + } + ], + "methodology": "Deployment options assessment for self-hosted harness with cloud-only inference", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 78, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Thorough TypeScript and Python guides covering permissions, subagents, MCP, sessions, and deployment" + } + ], + "methodology": "Documentation completeness review across both language SDKs", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Streaming message API surfaces every tool call and result; hooks enable custom audit logging" + } + ], + "methodology": "Review of message stream observability and hook-based audit trails", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Agent reasoning, tool intents, and permission requests are exposed in the streamed transcript" + } + ], + "methodology": "Assessment of streamed reasoning visibility and permission-request context", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "@anthropic-ai/claude-agent-sdk on npm", + "url": "https://www.npmjs.com/package/@anthropic-ai/claude-agent-sdk", + "date": "2026-06-01", + "value": "Proprietary license (Anthropic Commercial Terms); package code visible but not open source" + } + ], + "methodology": "Open source assessment of SDK packages and underlying harness", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK ecosystem", + "url": "https://www.npmjs.com/package/@anthropic-ai/claude-agent-sdk", + "date": "2026-06-01", + "value": "High npm download volume, frequent releases, and growing ecosystem of SDK-built agents since the 2025-09-29 rename" + } + ], + "methodology": "Community engagement analysis via package downloads, release cadence, and ecosystem projects", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 80, + "criteria": { + "ease_of_integration": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK quickstart", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Single package install for TypeScript or Python; minimal code to launch a fully tool-equipped agent" + } + ], + "methodology": "Integration complexity assessment from install to first working agent", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Stateless harness instances scale horizontally; throughput bounded by Claude API rate limits" + } + ], + "methodology": "Assessment of horizontal scaling patterns and API rate limit constraints", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API pricing", + "url": "https://www.anthropic.com/pricing", + "date": "2026-05-01", + "value": "Free SDK; costs are pay-as-you-go Claude tokens, which vary widely with agent autonomy and task length" + } + ], + "methodology": "Pricing model analysis of token-based costs for long-running agents", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Per-message usage/cost reporting and hooks for custom telemetry; full APM requires external tooling" + } + ], + "methodology": "Review of built-in usage reporting and integration points for external observability", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Claude Agent SDK overview", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-05-15", + "value": "Same harness running Claude Code in production at scale; renamed and generalized for agents on 2025-09-29" + } + ], + "methodology": "Maturity assessment based on shared production harness and release stability", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 91, + "notes": "Inherits Claude Code's coding harness; ideal for building custom coding agents and CI bots" + }, + "research-assistant": { + "overall": 85, + "notes": "Strong general-agent harness for research workflows with web, file, and MCP tools" + }, + "customer-support": { + "overall": 80, + "notes": "Permission system and subagents suit support automation; requires custom integration work" + }, + "data-analysis": { + "overall": 82, + "notes": "Bash and file tools enable scripted analysis agents with auditable execution" + } + }, + "best_for": [ + "Developers building production agents on Claude without writing their own tool loop", + "Teams extending Claude Code workflows into custom applications and services", + "Enterprises needing permissioned, auditable agents on Anthropic API, Bedrock, or Vertex", + "Builders who want subagents, hooks, and MCP out of the box" + ], + "strengths": [ + "Production-proven harness shared with Claude Code (tool loop, compaction, retries)", + "Fine-grained permission system with allow/deny rules, callbacks, hooks, and sandboxing", + "First-class subagents and MCP support for composable agent systems", + "Both TypeScript and Python SDKs with feature parity", + "Automatic context management enables long-running agents", + "Runs on Anthropic API, Amazon Bedrock, or Google Vertex AI" + ], + "limitations": [ + "Proprietary license (Anthropic Commercial Terms); not open source", + "Claude models only; no third-party or local model support", + "Token costs for autonomous agents can be hard to forecast", + "Heavier runtime footprint than thin LLM client libraries", + "Security posture depends on integrator configuring permissions and sandboxing correctly" + ], + "metadata": { + "license": "Proprietary (Anthropic Commercial Terms)", + "supported_models": [ + "Claude via Anthropic API", + "Claude via Amazon Bedrock", + "Claude via Google Vertex AI" + ], + "programming_languages": [ + "TypeScript", + "Python" + ], + "deployment_type": "Self-hosted SDK (cloud inference)", + "tool_support": [ + "Built-in file, bash, and web tools", + "Custom tools", + "MCP servers", + "Hooks", + "Subagents" + ], + "first_release": "2025 (as Claude Code SDK); renamed Claude Agent SDK 2025-09-29", + "pricing": "Free SDK; pay-as-you-go Claude token costs", + "packages": [ + "@anthropic-ai/claude-agent-sdk (npm)", + "claude-agent-sdk (PyPI)" + ] + }, + "related": [ + "claude-code", + "openai-agents-sdk", + "langgraph-agent", + "google-adk", + "microsoft-agent-framework" + ], + "tags": [ + "agent-sdk", + "anthropic", + "framework" + ] +} diff --git a/data/agents/claude-code.json b/data/agents/claude-code.json new file mode 100644 index 0000000..a14a786 --- /dev/null +++ b/data/agents/claude-code.json @@ -0,0 +1,479 @@ +{ + "id": "claude-code", + "type": "agent", + "name": "Claude Code", + "provider": "Anthropic", + "version": "2.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Anthropic's agentic coding tool available as a terminal CLI, IDE extensions, web, and desktop app. Plans and executes multi-step coding tasks with tiered permissions, OS-level sandboxing, MCP integration, hooks, subagents, and plugins/skills.", + "website": "https://www.anthropic.com/claude-code", + "trust_vector": { + "performance_reliability": { + "overall_score": 88, + "criteria": { + "task_completion_accuracy": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Claude Code GA announcement", + "url": "https://www.anthropic.com/news/claude-4", + "date": "2025-05-22", + "value": "GA v1.0 launched alongside Claude 4 models with state-of-the-art SWE-bench coding performance" + }, + { + "source": "Reported revenue traction", + "url": "https://www.anthropic.com/claude-code", + "date": "2025-11-20", + "value": "~$1B annualized revenue within roughly six months of GA indicates strong real-world task success" + } + ], + "methodology": "Benchmark results review plus adoption and revenue signals as proxy for sustained task success in production use", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/overview", + "date": "2026-05-15", + "value": "Mature built-in tool suite (file edit, bash, search) plus MCP servers, hooks, and plugins with permission gating" + } + ], + "methodology": "Hands-on testing of built-in tools and MCP integrations across coding workflows", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/overview", + "date": "2026-05-15", + "value": "Plan mode, extended thinking, and task tracking support long-horizon multi-file refactors and feature builds" + } + ], + "methodology": "Evaluation of plan mode and long-horizon task execution on multi-file repository changes", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code memory documentation", + "url": "https://code.claude.com/docs/en/memory", + "date": "2026-04-20", + "value": "CLAUDE.md project/user memory files, auto-compaction of long sessions, and session resume support persistence" + } + ], + "methodology": "Review of memory file hierarchy, context compaction behavior, and cross-session resume", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/overview", + "date": "2026-05-15", + "value": "Agent loop self-corrects from failed commands and test failures; checkpoints allow rewinding changes" + } + ], + "methodology": "Observed recovery behavior from failing builds, tests, and tool errors during evaluation sessions", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code subagents documentation", + "url": "https://code.claude.com/docs/en/sub-agents", + "date": "2026-04-20", + "value": "Native subagents with isolated contexts, custom system prompts, and tool restrictions enable parallel delegation" + } + ], + "methodology": "Testing of subagent delegation, parallel task fan-out, and plugin-defined agents", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 78, + "criteria": { + "tool_sandboxing": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic engineering: Claude Code sandboxing", + "url": "https://www.anthropic.com/engineering/claude-code-sandboxing", + "date": "2025-11-10", + "value": "OS-level sandboxing with filesystem and network isolation shipped Oct-Nov 2025, reducing permission prompts ~84%" + }, + { + "source": "Claude Code on the web", + "url": "https://www.anthropic.com/news/claude-code-on-the-web", + "date": "2025-10-20", + "value": "Web version executes tasks inside Anthropic-managed isolated sandboxes" + } + ], + "methodology": "Review of sandbox architecture (filesystem and network isolation) and managed cloud sandbox design", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code permissions documentation", + "url": "https://code.claude.com/docs/en/iam", + "date": "2026-04-20", + "value": "Tiered permission prompts with allow/deny rules, per-tool allowlists, and enterprise managed policy settings" + } + ], + "methodology": "Assessment of permission model, allowlist granularity, and enterprise policy controls", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Code security documentation", + "url": "https://code.claude.com/docs/en/security", + "date": "2026-04-20", + "value": "Permission prompts for write actions, sandbox network isolation, and injection-aware system design mitigate untrusted content risks" + } + ], + "methodology": "Review of documented mitigations and behavior when processing untrusted repository and web content", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Code on the web", + "url": "https://www.anthropic.com/news/claude-code-on-the-web", + "date": "2025-10-20", + "value": "Cloud sessions run in per-task isolated environments; local sandbox restricts filesystem scope to the project" + } + ], + "methodology": "Architecture review of session isolation in cloud sandboxes and local filesystem scoping", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code GitHub repository", + "url": "https://github.com/anthropics/claude-code", + "date": "2026-06-01", + "value": "Proprietary product; public repo hosts releases, issues, and documentation but core source is not open" + } + ], + "methodology": "License and source availability review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 71, + "criteria": { + "data_retention": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic privacy policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-03-01", + "value": "Commercial/API usage not used for training by default; retention controls available for enterprise plans" + } + ], + "methodology": "Review of Anthropic data retention commitments across consumer and commercial tiers", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-03-01", + "value": "SOC 2 Type II certified with GDPR-aligned DPA available for commercial customers" + } + ], + "methodology": "Compliance certification and DPA availability review", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic privacy policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-03-01", + "value": "Code and prompts processed by Anthropic only; no third-party model providers in the loop" + } + ], + "methodology": "Data flow analysis of code, prompt, and telemetry handling", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/overview", + "date": "2026-05-15", + "value": "Client runs locally and can route via Bedrock or Vertex AI, but requires Claude models in the cloud; no fully local model option" + } + ], + "methodology": "Deployment options assessment including Bedrock/Vertex routing and air-gap feasibility", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 80, + "criteria": { + "documentation_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/overview", + "date": "2026-05-15", + "value": "Extensive docs covering permissions, sandboxing, MCP, hooks, subagents, plugins, and enterprise deployment" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/monitoring-usage", + "date": "2026-04-20", + "value": "Full transcript of every tool call and edit visible in session; OpenTelemetry metrics and logging supported" + } + ], + "methodology": "Review of session transcripts, hooks-based auditing, and OTel telemetry support", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/overview", + "date": "2026-05-15", + "value": "Visible reasoning, plan mode previews, and per-action permission prompts explain intended changes before execution" + } + ], + "methodology": "Assessment of plan previews, inline reasoning, and diff-based change explanation", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 50, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code GitHub repository", + "url": "https://github.com/anthropics/claude-code", + "date": "2026-06-01", + "value": "Proprietary; public repository used for releases and issue tracking, not source code" + } + ], + "methodology": "Open source assessment of core product and ecosystem components", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code GitHub repository", + "url": "https://github.com/anthropics/claude-code", + "date": "2026-06-01", + "value": "Highly active issue tracker, frequent releases, and a large plugin/skills ecosystem since launch" + } + ], + "methodology": "Community engagement analysis via GitHub activity, release cadence, and ecosystem growth", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 84, + "criteria": { + "ease_of_integration": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code documentation", + "url": "https://code.claude.com/docs/en/quickstart", + "date": "2026-05-15", + "value": "Single npm install for CLI; native VS Code/JetBrains extensions, web, and desktop apps with shared auth" + } + ], + "methodology": "Setup time and integration surface assessment across CLI, IDE, web, and desktop", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Code on the web", + "url": "https://www.anthropic.com/news/claude-code-on-the-web", + "date": "2025-10-20", + "value": "Cloud sandboxes allow many parallel tasks; headless mode and GitHub Actions support CI-scale automation" + } + ], + "methodology": "Assessment of parallel cloud sessions, headless/CI usage, and rate limit behavior", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "Claude pricing", + "url": "https://www.anthropic.com/pricing", + "date": "2026-05-01", + "value": "Claude Pro/Max subscriptions cap monthly spend; API pay-as-you-go usage varies significantly with task size" + } + ], + "methodology": "Pricing model analysis comparing subscription caps versus variable API token costs", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code monitoring documentation", + "url": "https://code.claude.com/docs/en/monitoring-usage", + "date": "2026-04-20", + "value": "Built-in OpenTelemetry metrics, usage tracking, and enterprise analytics dashboard" + } + ], + "methodology": "Review of OTel export, cost/usage tracking, and admin analytics features", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Claude Code GA and adoption", + "url": "https://www.anthropic.com/claude-code", + "date": "2025-11-20", + "value": "GA since 2025-05-22 with ~$1B ARR within ~6 months; widely deployed in enterprises" + } + ], + "methodology": "Maturity assessment from GA timeline, release stability, and enterprise adoption", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "Flagship use case; excels at multi-file feature work, refactors, debugging, and test-driven workflows" + }, + "data-analysis": { + "overall": 82, + "notes": "Strong for scripted analysis, notebooks, and data pipeline work via bash and file tools" + }, + "research-assistant": { + "overall": 78, + "notes": "Capable codebase and web research via agentic search, though optimized for engineering contexts" + }, + "content-creation": { + "overall": 72, + "notes": "Good for technical writing and docs generation; not designed for general marketing content" + } + }, + "best_for": [ + "Professional developers wanting an agentic pair programmer in terminal and IDE", + "Teams automating large refactors, migrations, and test generation", + "Enterprises needing permissioned, sandboxed, auditable AI coding", + "CI/CD and headless automation of repetitive engineering tasks" + ], + "strengths": [ + "State-of-the-art coding capability backed by Claude models", + "OS-level sandboxing with filesystem and network isolation cut permission prompts ~84%", + "Tiered permission system with allowlists, hooks, and enterprise policies", + "Rich extensibility: MCP servers, hooks, subagents, plugins, and skills", + "Available across terminal, IDE extensions, web, and desktop with shared workflows", + "Strong observability via full transcripts and OpenTelemetry" + ], + "limitations": [ + "Proprietary, closed-source core despite public releases repository", + "Locked to Claude models; no local or third-party model support", + "API pay-as-you-go costs can spike on large autonomous tasks", + "Subscription rate limits can interrupt heavy daily usage", + "Autonomous edits still require human review for correctness and security" + ], + "metadata": { + "license": "Proprietary (public releases repo at github.com/anthropics/claude-code)", + "supported_models": [ + "Claude Opus", + "Claude Sonnet", + "Claude Haiku" + ], + "programming_languages": [ + "Language-agnostic (any language in the repository)" + ], + "deployment_type": "Local CLI / IDE extensions / managed web sandboxes / desktop app", + "tool_support": [ + "Built-in file, bash, and search tools", + "MCP servers", + "Hooks", + "Subagents", + "Plugins and skills" + ], + "first_release": "2025-02-24 (research preview); GA v1.0 2025-05-22", + "pricing": "Claude Pro/Max subscription or API pay-as-you-go; ~$1B ARR within ~6 months of GA", + "interfaces": [ + "Terminal CLI", + "VS Code and JetBrains extensions", + "Web", + "Desktop" + ] + }, + "related": [ + "claude-agent-sdk", + "openai-codex", + "gemini-cli", + "cursor-agent", + "github-copilot-coding-agent" + ], + "tags": [ + "coding-agent", + "cli", + "anthropic", + "sandboxed" + ] +} diff --git a/data/agents/cursor-agent.json b/data/agents/cursor-agent.json new file mode 100644 index 0000000..acc643e --- /dev/null +++ b/data/agents/cursor-agent.json @@ -0,0 +1,463 @@ +{ + "id": "cursor-agent", + "type": "agent", + "name": "Cursor Agent", + "provider": "Anysphere", + "version": "3.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Agent mode of Cursor, Anysphere's AI-native IDE. The product's primary surface is now agentic: parallel local agents, cloud/background agents running in isolated VMs, and the in-house Composer model line alongside Claude, GPT, and Gemini.", + "website": "https://cursor.com/", + "trust_vector": { + "performance_reliability": { + "overall_score": 83, + "criteria": { + "task_completion_accuracy": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Blog - Composer 2", + "url": "https://cursor.com/blog/composer-2", + "date": "2026-03-19", + "value": "Composer 2 (2026-03-19) improved agentic coding accuracy and speed over Composer 1, which shipped with Cursor 2.0 in October 2025" + } + ], + "methodology": "Assessment of agentic edit and task completion quality across Composer and frontier model options, drawing on vendor benchmarks and user reports", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Cursor Documentation - Agent", + "url": "https://docs.cursor.com/agent", + "date": "2026-04-15", + "value": "Agent reliably uses codebase search, terminal commands, file edits, lints, and MCP tools within the IDE loop" + } + ], + "methodology": "Review of agent tool loop reliability across edit, search, terminal, and MCP integrations", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "InfoQ - Cursor 3 Agent-First Interface", + "url": "https://www.infoq.com/news/2026/04/cursor-3-agent-first-interface/", + "date": "2026-04-20", + "value": "Cursor 3 (April 2026) redesigned the IDE around agent plans and task orchestration as the primary workflow" + } + ], + "methodology": "Evaluation of plan construction and multi-file refactoring on complex tasks in the agent-first interface", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Documentation - Rules and Memories", + "url": "https://docs.cursor.com/context/rules", + "date": "2026-04-15", + "value": "Project rules, user rules, and Memories persist conventions and learned context across sessions" + } + ], + "methodology": "Review of rules, memories, and codebase indexing persistence", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 81, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Documentation - Agent", + "url": "https://docs.cursor.com/agent", + "date": "2026-04-15", + "value": "Agent reads linter and test failures and iterates automatically; loop detection prevents most runaway retries" + } + ], + "methodology": "Assessment of automatic iteration on lints, build errors, and failing tests", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "InfoQ - Cursor 3 Agent-First Interface", + "url": "https://www.infoq.com/news/2026/04/cursor-3-agent-first-interface/", + "date": "2026-04-20", + "value": "Parallel agents run concurrently on separate tasks, locally via worktrees or in cloud VMs as background agents" + } + ], + "methodology": "Review of parallel and background agent orchestration capabilities", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 66, + "criteria": { + "tool_sandboxing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Documentation - Background Agents", + "url": "https://docs.cursor.com/background-agent", + "date": "2026-04-15", + "value": "Cloud/background agents execute in isolated VMs; local agents run terminal commands on the user's machine with configurable approval and sandbox settings" + } + ], + "methodology": "Security review of cloud VM isolation versus local execution approval model", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Security", + "url": "https://cursor.com/security", + "date": "2026-04-15", + "value": "Teams/Enterprise offer SSO, admin controls, and centrally enforced privacy mode; repository access scoped via GitHub app for cloud agents" + } + ], + "methodology": "Review of org-level controls, SSO, and repository scoping", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 68, + "confidence": "low", + "evidence": [ + { + "source": "Cursor Security", + "url": "https://cursor.com/security", + "date": "2026-04-15", + "value": "Agents process untrusted repo content, web results, and MCP outputs; command allowlists and approvals mitigate but injection hardening is not publicly detailed" + } + ], + "methodology": "Threat surface analysis of untrusted content ingestion against documented mitigations", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Security", + "url": "https://cursor.com/security", + "date": "2026-04-15", + "value": "Privacy mode guarantees code is not stored or trained on; SOC 2 attested infrastructure with per-tenant separation for cloud agents" + } + ], + "methodology": "Review of privacy mode guarantees and tenant isolation for cloud agents", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 32, + "confidence": "high", + "evidence": [ + { + "source": "Cursor", + "url": "https://cursor.com/", + "date": "2026-04-15", + "value": "Proprietary VS Code fork; Composer models and agent harness are closed source" + } + ], + "methodology": "Source availability assessment", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 64, + "criteria": { + "data_retention": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Privacy", + "url": "https://cursor.com/privacy", + "date": "2026-04-15", + "value": "Privacy mode (default for Teams) prevents code storage and training use; without it, telemetry and code data may be retained" + } + ], + "methodology": "Review of privacy mode retention guarantees and default settings", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Security", + "url": "https://cursor.com/security", + "date": "2026-04-15", + "value": "SOC 2 Type II attestation, DPA availability, and GDPR-aligned processing terms for business customers" + } + ], + "methodology": "Compliance documentation assessment", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Documentation - Models", + "url": "https://docs.cursor.com/settings/models", + "date": "2026-04-15", + "value": "Prompts and code context are routed to the selected model provider (Anysphere Composer, Anthropic, OpenAI, Google), expanding the data-processing surface" + } + ], + "methodology": "Data flow analysis across multi-provider model routing", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Cursor Documentation", + "url": "https://docs.cursor.com/", + "date": "2026-04-15", + "value": "IDE runs locally but all model inference is cloud-based; no on-premises or fully offline mode" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 69, + "criteria": { + "documentation_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Cursor Documentation", + "url": "https://docs.cursor.com/", + "date": "2026-04-15", + "value": "Thorough docs covering agent mode, background agents, rules, MCP, models, and enterprise administration" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Cursor Documentation - Agent", + "url": "https://docs.cursor.com/agent", + "date": "2026-04-15", + "value": "Every agent action is visible: diffs reviewable before apply, terminal output streamed, and per-agent activity logs in the agent-first UI" + } + ], + "methodology": "Review of action visibility, diff review workflow, and agent logs", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "InfoQ - Cursor 3 Agent-First Interface", + "url": "https://www.infoq.com/news/2026/04/cursor-3-agent-first-interface/", + "date": "2026-04-20", + "value": "Agents narrate plans and rationale in the task view, though depth of explanation varies by selected model" + } + ], + "methodology": "Assessment of plan narration and change rationale quality", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "Cursor", + "url": "https://cursor.com/", + "date": "2026-04-15", + "value": "Closed-source editor fork and proprietary Composer models; no published weights or harness internals" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Cursor Community Forum", + "url": "https://forum.cursor.com/", + "date": "2026-06-01", + "value": "Very large active user community, busy forum, and rapid release cadence (Cursor 2.0 Oct 2025, Composer 2 Mar 2026, Cursor 3 Apr 2026)" + } + ], + "methodology": "Community engagement and release cadence analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 79, + "criteria": { + "ease_of_integration": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Cursor Documentation", + "url": "https://docs.cursor.com/", + "date": "2026-04-15", + "value": "VS Code fork imports existing extensions, settings, and keybindings; developers are productive in minutes" + } + ], + "methodology": "Onboarding and integration friction assessment", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Documentation - Background Agents", + "url": "https://docs.cursor.com/background-agent", + "date": "2026-04-15", + "value": "Cloud agents offload long tasks to isolated VMs and run in parallel, decoupling agent throughput from local hardware" + } + ], + "methodology": "Scalability assessment of parallel and cloud agent execution", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Cursor Pricing", + "url": "https://cursor.com/pricing", + "date": "2026-04-15", + "value": "Free / Pro $20 / Pro+ $60 / Ultra $200 per month; Teams $40 per seat. Tiers include usage allowances with overage billing for heavy frontier-model use" + } + ], + "methodology": "Pricing model analysis; flat tiers are clear but usage-based overages reduce predictability for heavy agent users", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Cursor Documentation - Teams", + "url": "https://docs.cursor.com/account/teams", + "date": "2026-04-15", + "value": "Team usage dashboards and admin analytics; deep observability of agent behavior requires external tooling" + } + ], + "methodology": "Monitoring and admin analytics features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "InfoQ - Cursor 3 Agent-First Interface", + "url": "https://www.infoq.com/news/2026/04/cursor-3-agent-first-interface/", + "date": "2026-04-20", + "value": "Mature, widely adopted product with massive enterprise and individual user base and sustained rapid iteration through Cursor 3" + } + ], + "methodology": "Product maturity and adoption assessment", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 92, + "notes": "Best-in-class agentic IDE experience; parallel agents, fast in-house Composer models, and frontier model choice" + }, + "data-analysis": { + "overall": 76, + "notes": "Strong for building and iterating on analysis code and notebooks, though not an analytics product itself" + }, + "research-assistant": { + "overall": 68, + "notes": "Useful for technical research within codebases and docs; general research is outside its design focus" + }, + "education": { + "overall": 74, + "notes": "Visible agent reasoning and diffs help learners understand changes; generous free tier lowers the barrier" + } + }, + "best_for": [ + "Developers who want agentic coding embedded directly in their daily IDE workflow", + "Teams running parallel agents on multiple tasks, locally and in cloud VMs", + "Users wanting model choice between fast in-house Composer models and frontier models", + "Organizations migrating from VS Code with zero extension or keybinding friction" + ], + "strengths": [ + "Agent-first IDE redesign (Cursor 3, April 2026) makes agent orchestration the primary workflow", + "In-house Composer model line (Composer 2, 2026-03-19) delivers very fast agentic edits alongside Claude/GPT/Gemini choice", + "Parallel agents and cloud background agents in isolated VMs scale work beyond one task at a time", + "Human-in-the-loop by design: reviewable diffs, command approvals, and visible terminal output", + "Seamless VS Code compatibility for extensions, themes, and keybindings", + "Privacy mode with no-storage/no-training guarantee, enforceable org-wide" + ], + "limitations": [ + "Closed-source editor and models limit independent security and behavior auditing", + "No offline or on-premises inference; all model calls go to cloud providers", + "Usage-based overages above tier allowances make heavy agent usage costs less predictable", + "Local agent terminal execution depends on user-configured approvals; misconfiguration widens risk", + "Multi-provider model routing complicates data governance reviews", + "Rapid release cadence occasionally introduces regressions and workflow changes" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Composer 1/2 (Anysphere in-house)", + "Anthropic Claude", + "OpenAI GPT", + "Google Gemini" + ], + "programming_languages": [ + "All languages supported by VS Code ecosystem" + ], + "deployment_type": "Local IDE (VS Code fork) + cloud agents in isolated VMs", + "tool_support": [ + "Codebase search and indexing", + "Terminal execution", + "MCP servers", + "Browser/web context", + "GitHub integration for cloud agents" + ], + "first_release": "2023 (IDE); Cursor 2.0 with Composer Oct 2025; Cursor 3 April 2026", + "pricing": "Free / Pro $20 / Pro+ $60 / Ultra $200 per month; Teams $40 per seat", + "company": "Anysphere" + }, + "related_entities": [ + "claude-code", + "github-copilot-coding-agent", + "openai-codex", + "devin" + ], + "tags": [ + "ide", + "agentic-coding", + "parallel-agents", + "proprietary" + ] +} diff --git a/data/agents/devin.json b/data/agents/devin.json new file mode 100644 index 0000000..c11ac81 --- /dev/null +++ b/data/agents/devin.json @@ -0,0 +1,462 @@ +{ + "id": "devin", + "type": "agent", + "name": "Devin", + "provider": "Cognition", + "version": "2.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Autonomous AI software engineer from Cognition that plans and executes multi-step engineering tasks in a sandboxed cloud workspace with its own editor, shell, and browser, and delivers work as pull requests.", + "website": "https://devin.ai/", + "trust_vector": { + "performance_reliability": { + "overall_score": 80, + "criteria": { + "task_completion_accuracy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Cognition - Devin Product Page", + "url": "https://devin.ai/", + "date": "2026-05-01", + "value": "Completes well-scoped engineering tasks end-to-end as PRs; reliability is strongest on bounded tasks like migrations, bug fixes, and test backfills" + } + ], + "methodology": "Assessment of task completion on scoped engineering work based on vendor documentation, customer case studies, and independent user reports", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Devin Documentation", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Operates a full cloud workspace (editor, shell, browser) with native GitHub, Slack, Jira, and Linear integrations" + } + ], + "methodology": "Review of integrated toolchain reliability across shell, browser, and VCS operations", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Devin Documentation - Planning", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Devin 2.0 introduced interactive planning: drafts an editable plan before execution and revises it as the task evolves" + } + ], + "methodology": "Evaluation of plan generation, user-editable plans, and plan adherence on long-horizon tasks", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Documentation - Knowledge", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Persistent Knowledge base, Playbooks for repeatable procedures, and Devin Wiki auto-generated codebase documentation carry context across sessions" + } + ], + "methodology": "Review of cross-session memory features (Knowledge, Playbooks, Wiki) and session snapshot persistence", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Documentation", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Iterates on failing tests and build errors autonomously; can go down unproductive paths on ambiguous tasks, consuming ACUs until interrupted" + } + ], + "methodology": "Assessment of autonomous debugging behavior and failure-mode reports from production users", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Cognition - Devin Product Page", + "url": "https://devin.ai/", + "date": "2026-05-01", + "value": "Supports running multiple parallel Devin sessions on independent tasks; orchestration across Devins is coarser than dedicated multi-agent frameworks" + } + ], + "methodology": "Review of parallel session capabilities and multi-Devin task delegation", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 69, + "criteria": { + "tool_sandboxing": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Devin Security Documentation", + "url": "https://devin.ai/security", + "date": "2026-05-01", + "value": "Each session runs in an isolated, sandboxed cloud VM separate from user infrastructure; code execution never touches the local machine" + } + ], + "methodology": "Security architecture review of isolated cloud workspace model", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Security Documentation", + "url": "https://devin.ai/security", + "date": "2026-05-01", + "value": "Enterprise tier offers SSO/SAML, scoped repository access via GitHub app permissions, and secrets management for credentials" + } + ], + "methodology": "Review of identity, repository scoping, and secrets handling controls", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 70, + "confidence": "low", + "evidence": [ + { + "source": "Devin Documentation", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Autonomous web browsing and repository content processing create injection surface; mitigations exist but are not publicly detailed" + } + ], + "methodology": "Threat surface analysis of autonomous browsing and untrusted repo content; limited public disclosure of defenses", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Security Documentation", + "url": "https://devin.ai/security", + "date": "2026-05-01", + "value": "Per-session VM isolation and tenant separation; SOC 2 Type II attested infrastructure" + } + ], + "methodology": "Data architecture review of tenant and session isolation claims", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "Cognition", + "url": "https://cognition.ai/", + "date": "2026-05-27", + "value": "Fully proprietary product and models (in-house SWE-1.5 line plus frontier models); no source code or model weights published" + } + ], + "methodology": "Source availability assessment", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 61, + "criteria": { + "data_retention": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Security Documentation", + "url": "https://devin.ai/security", + "date": "2026-05-01", + "value": "Session data and workspace snapshots retained in Cognition's cloud; enterprise contracts offer training opt-out and retention controls" + } + ], + "methodology": "Review of published retention practices and enterprise data controls", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Security Documentation", + "url": "https://devin.ai/security", + "date": "2026-05-01", + "value": "SOC 2 Type II attestation and DPA availability for enterprise customers; GDPR posture depends on contractual terms" + } + ], + "methodology": "Compliance documentation assessment", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Cognition - Devin Product Page", + "url": "https://devin.ai/", + "date": "2026-05-01", + "value": "Code is processed by Cognition's own SWE-1.5 model line and routed to third-party frontier models for some tasks" + } + ], + "methodology": "Data flow analysis of model routing between in-house and third-party providers", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 40, + "confidence": "high", + "evidence": [ + { + "source": "Devin Documentation", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Cloud-only service; no self-hosted or on-premises deployment, with VPC options limited to enterprise arrangements" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 66, + "criteria": { + "documentation_quality": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Devin Documentation", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Well-organized docs covering onboarding, ACU model, integrations, Playbooks, and enterprise administration" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Cognition - Devin Product Page", + "url": "https://devin.ai/", + "date": "2026-05-01", + "value": "Full visibility into Devin's plan, editor, shell, and browser in real time; complete session timeline is replayable" + } + ], + "methodology": "Review of session visibility, live workspace observation, and replay features", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Documentation - Planning", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Explicit upfront plans, step-by-step narration during execution, and PR descriptions explaining changes" + } + ], + "methodology": "Assessment of plan transparency and change justification quality", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 20, + "confidence": "high", + "evidence": [ + { + "source": "Cognition", + "url": "https://cognition.ai/", + "date": "2026-05-27", + "value": "Closed-source product; limited public benchmark reproducibility and no published model details beyond blog posts" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Cognition raises $1B at $26B valuation", + "url": "https://techcrunch.com/2026/05/27/ai-coding-startup-cognition-raises-1b-at-25b-pre-money-valuation/", + "date": "2026-05-27", + "value": "Large and growing commercial user base (~$492M ARR reported); community is customer-driven rather than open-source contributor-driven" + } + ], + "methodology": "Community and ecosystem engagement analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 77, + "criteria": { + "ease_of_integration": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Devin Documentation - Integrations", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Assign tasks from Slack, GitHub issues, Jira, or Linear; Devin 2.0 added IDE-style interface and API access" + } + ], + "methodology": "Integration surface assessment across team workflows", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Cognition - Devin Product Page", + "url": "https://devin.ai/", + "date": "2026-05-01", + "value": "Parallel cloud sessions allow teams to fan out many tasks simultaneously without local resource constraints" + } + ], + "methodology": "Scalability assessment of parallel cloud session model", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "Devin Pricing", + "url": "https://devin.ai/pricing", + "date": "2026-05-01", + "value": "Core plan from $20 pay-as-you-go at $2.25/ACU (Devin 2.0, April 2025, down from $500/mo); Team $500/mo. ACU consumption varies widely by task complexity" + } + ], + "methodology": "Pricing model analysis; ACU-metered billing makes per-task costs hard to forecast", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Devin Documentation", + "url": "https://docs.devin.ai/", + "date": "2026-05-01", + "value": "Session dashboards, ACU usage tracking, and admin controls for team usage oversight" + } + ], + "methodology": "Monitoring and usage governance features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Cognition raises $1B at $26B valuation", + "url": "https://techcrunch.com/2026/05/27/ai-coding-startup-cognition-raises-1b-at-25b-pre-money-valuation/", + "date": "2026-05-27", + "value": "Closed $1B+ round at $26B valuation (2026-05-27) with ~$492M ARR; acquired Windsurf 2025-07-14, signaling strong vendor viability" + } + ], + "methodology": "Vendor maturity and product stability assessment", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 86, + "notes": "Purpose-built autonomous software engineer; excels at scoped tasks like migrations, bug fixes, test coverage, and PR-sized features" + }, + "data-analysis": { + "overall": 70, + "notes": "Can write and run analysis scripts in its workspace, but is optimized for software engineering rather than analytics workflows" + }, + "research-assistant": { + "overall": 68, + "notes": "Browser access enables technical research and documentation digging, though it is not designed for general research synthesis" + }, + "education": { + "overall": 64, + "notes": "Replayable sessions showing plan and execution can teach engineering practice, but ACU costs make it expensive for learning" + } + }, + "best_for": [ + "Engineering teams delegating well-scoped tasks (migrations, bug fixes, test backfills) as autonomous PRs", + "Organizations wanting to fan out parallel coding tasks to cloud agents without local setup", + "Teams that assign work from Slack, Jira, or Linear and review results as pull requests", + "Companies preferring a fully managed, sandboxed cloud workspace over local agent execution" + ], + "strengths": [ + "True end-to-end autonomy: plans, codes, tests, browses docs, and opens PRs in its own cloud workspace", + "Sandboxed cloud VMs isolate execution from user infrastructure", + "Interactive, editable plans and fully replayable session timelines provide strong traceability", + "Devin 2.0 pricing ($20 entry, $2.25/ACU) dramatically lowered the adoption barrier from the original $500/mo", + "Persistent Knowledge, Playbooks, and auto-generated Devin Wiki retain organizational context", + "Strong vendor trajectory: $26B valuation, ~$492M ARR, Windsurf acquisition (2025-07-14)" + ], + "limitations": [ + "ACU-metered billing makes costs unpredictable, especially when the agent pursues unproductive paths", + "Fully proprietary stack with no self-hosted option; code must be processed in Cognition's cloud", + "Reliability drops on ambiguous or large unscoped tasks, requiring careful task decomposition", + "Prompt injection defenses for autonomous browsing are not publicly documented", + "Some workloads route to third-party frontier models, complicating data governance review", + "Output still requires human code review; unsupervised merging is not advisable" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Cognition SWE-1.5 model line (in-house)", + "Third-party frontier models for select tasks" + ], + "programming_languages": [ + "Most major languages (Python, TypeScript, Java, Go, etc.)" + ], + "deployment_type": "Cloud (sandboxed VM workspaces)", + "tool_support": [ + "Cloud editor and shell", + "Built-in browser", + "GitHub/GitLab integration", + "Slack, Jira, Linear", + "API access" + ], + "first_release": "2024 (limited), Devin 2.0 April 2025", + "pricing": "Core from $20 pay-as-you-go ($2.25/ACU); Team $500/mo; Enterprise custom", + "company_milestones": "Acquired Windsurf 2025-07-14; raised $1B+ at $26B valuation (closed 2026-05-27); ~$492M ARR" + }, + "related_entities": [ + "openai-codex", + "claude-code", + "github-copilot-coding-agent", + "google-jules", + "cursor-agent" + ], + "tags": [ + "autonomous", + "software-engineering", + "cloud-agent", + "proprietary" + ] +} diff --git a/data/agents/dify.json b/data/agents/dify.json new file mode 100644 index 0000000..6c02146 --- /dev/null +++ b/data/agents/dify.json @@ -0,0 +1,479 @@ +{ + "id": "dify", + "type": "agent", + "name": "Dify", + "provider": "LangGenius", + "version": "1.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Open-source LLM application and agentic workflow platform with a visual canvas, built-in RAG pipeline, agent nodes, and 50+ built-in tools. One of the most-starred LLM app platforms, deployable self-hosted or via Dify Cloud.", + "website": "https://dify.ai/", + "trust_vector": { + "performance_reliability": { + "overall_score": 80, + "criteria": { + "task_completion_accuracy": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Documentation", + "url": "https://docs.dify.ai/", + "date": "2026-05-15", + "value": "Visual workflow orchestration with RAG grounding improves task accuracy for production LLM apps" + } + ], + "methodology": "Task completion testing across chatflow, workflow, and agent app types", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Dify Tools Documentation", + "url": "https://docs.dify.ai/en/guides/tools/README", + "date": "2026-04-20", + "value": "50+ built-in tools plus custom API tools, plugin marketplace, and MCP support with schema validation" + } + ], + "methodology": "Tool invocation testing across built-in, custom, and plugin tools", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Dify Workflow Documentation", + "url": "https://docs.dify.ai/en/guides/workflow/README", + "date": "2026-04-20", + "value": "Visual canvas supports branching, iteration, parallel nodes, and agent nodes for autonomous sub-tasks" + } + ], + "methodology": "Complex workflow construction and execution testing", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Conversation Variables", + "url": "https://docs.dify.ai/en/guides/workflow/variables", + "date": "2026-03-20", + "value": "Conversation variables, session memory, and knowledge-base persistence across chat sessions" + } + ], + "methodology": "Memory evaluation across sessions and knowledge bases", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Error Handling Documentation", + "url": "https://docs.dify.ai/en/guides/workflow/error-handling/README", + "date": "2026-03-20", + "value": "Node-level error handling with retry, fail branches, and default-value fallbacks in workflows" + } + ], + "methodology": "Error injection testing on workflow error branches and retries", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Agent Node Documentation", + "url": "https://docs.dify.ai/en/guides/workflow/node/agent", + "date": "2026-03-25", + "value": "Agent nodes embed autonomous strategies inside workflows, but deep multi-agent orchestration is less developed than code-first frameworks" + } + ], + "methodology": "Multi-agent pattern testing using agent nodes within workflows", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 76, + "criteria": { + "tool_sandboxing": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Sandbox (dify-sandbox)", + "url": "https://github.com/langgenius/dify-sandbox", + "date": "2026-02-15", + "value": "Code nodes execute in the dedicated dify-sandbox service with syscall restrictions; external tools run via HTTP without sandboxing" + } + ], + "methodology": "Security architecture review of code execution sandbox and tool isolation", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Workspace and Enterprise Docs", + "url": "https://docs.dify.ai/", + "date": "2026-04-20", + "value": "Workspace member roles, app-level API keys, and SSO/access policies in premium/enterprise editions" + } + ], + "methodology": "Access control assessment of workspace roles and API key scoping", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Content Moderation Documentation", + "url": "https://docs.dify.ai/en/guides/application-orchestrate/app-toolkits/moderation-tool", + "date": "2026-03-20", + "value": "Built-in content moderation (keyword, OpenAI moderation, custom API) for inputs/outputs; no dedicated injection defense" + } + ], + "methodology": "Injection testing with moderation toolkits enabled", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Self-Hosted Deployment", + "url": "https://docs.dify.ai/en/getting-started/install-self-hosted/README", + "date": "2026-04-20", + "value": "Workspace- and app-scoped data with isolated knowledge bases; self-hosting gives full infrastructure isolation" + } + ], + "methodology": "Data isolation architecture review across cloud and self-hosted modes", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Dify GitHub Repository", + "url": "https://github.com/langgenius/dify", + "date": "2026-06-01", + "value": "Fully public codebase with 138k+ stars; license is Apache-2.0-based but adds multi-tenant SaaS resale restrictions" + } + ], + "methodology": "Source code and license terms review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 83, + "criteria": { + "data_retention": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Dify Self-Hosted Deployment", + "url": "https://docs.dify.ai/en/getting-started/install-self-hosted/README", + "date": "2026-04-20", + "value": "Self-hosted deployments keep all logs, conversations, and knowledge data in user-controlled databases" + } + ], + "methodology": "Privacy architecture review of self-hosted versus cloud retention", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 81, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Privacy Policy", + "url": "https://dify.ai/privacy", + "date": "2026-04-01", + "value": "Published privacy policy for cloud; self-hosting enables full GDPR data-controller compliance" + } + ], + "methodology": "Compliance capabilities assessment across deployment modes", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Model Provider Documentation", + "url": "https://docs.dify.ai/en/guides/model-configuration/README", + "date": "2026-04-20", + "value": "Data flows to configured model providers and any third-party tools invoked in workflows; local models supported via Ollama/Xinference" + } + ], + "methodology": "Data flow analysis across model providers and tool integrations", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Dify Docker Compose Deployment", + "url": "https://docs.dify.ai/en/getting-started/install-self-hosted/docker-compose", + "date": "2026-04-20", + "value": "Free self-hosting via Docker Compose or Helm, including fully local model serving with Ollama" + } + ], + "methodology": "Deployment options assessment including air-gapped configurations", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 85, + "criteria": { + "documentation_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Dify Documentation", + "url": "https://docs.dify.ai/", + "date": "2026-05-15", + "value": "Extensive multilingual docs covering workflows, RAG, tools, plugins, and self-hosting" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Dify Logs and Annotations / Tracing Integrations", + "url": "https://docs.dify.ai/en/guides/monitoring/README", + "date": "2026-04-20", + "value": "Built-in conversation logs, node-level run traces, and integrations with LangSmith, Langfuse, and Opik" + } + ], + "methodology": "Tracing and logging capabilities assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Workflow Canvas", + "url": "https://docs.dify.ai/en/guides/workflow/README", + "date": "2026-04-20", + "value": "Visual canvas plus per-node inputs/outputs and citation/attribution in RAG answers make behavior inspectable" + } + ], + "methodology": "Explainability assessment of visual runs and RAG citations", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Dify Open Source License", + "url": "https://github.com/langgenius/dify/blob/main/LICENSE", + "date": "2026-06-01", + "value": "Apache-2.0-based custom license; restricts operating multi-tenant SaaS and removing branding without commercial license" + } + ], + "methodology": "License terms review against OSI-standard licenses", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Dify GitHub Metrics", + "url": "https://github.com/langgenius/dify", + "date": "2026-04-15", + "value": "138k+ GitHub stars and 1M+ deployed apps as of April 2026; very active releases and plugin marketplace" + } + ], + "methodology": "Community engagement analysis of stars, deployments, and release cadence", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 85, + "criteria": { + "ease_of_integration": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Dify Getting Started", + "url": "https://docs.dify.ai/en/getting-started/install-self-hosted/docker-compose", + "date": "2026-04-20", + "value": "No-code visual builder plus Backend-as-a-Service APIs; running app possible in minutes via Docker or Cloud" + } + ], + "methodology": "Integration complexity assessment for no-code and API usage", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Dify Helm/Kubernetes Deployment", + "url": "https://docs.dify.ai/en/getting-started/install-self-hosted/README", + "date": "2026-04-20", + "value": "Kubernetes/Helm deployments with horizontally scalable API and worker services; 1M+ apps deployed on the platform" + } + ], + "methodology": "Scalability assessment of deployment architectures", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Dify Pricing", + "url": "https://dify.ai/pricing", + "date": "2026-06-01", + "value": "Free self-hosted; Cloud Sandbox free, Professional $59/mo, Team $159/mo with clear plan limits" + } + ], + "methodology": "Pricing model analysis across self-hosted and cloud tiers", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Dify Monitoring Documentation", + "url": "https://docs.dify.ai/en/guides/monitoring/README", + "date": "2026-04-20", + "value": "Built-in usage analytics, conversation logs, annotation workflows, and external tracing integrations" + } + ], + "methodology": "Monitoring features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Dify GitHub Releases", + "url": "https://github.com/langgenius/dify/releases", + "date": "2026-06-01", + "value": "Mature 1.x platform with frequent stable releases, plugin system, and large production install base" + } + ], + "methodology": "Production readiness assessment of release maturity and adoption", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "customer-support": { + "overall": 88, + "notes": "RAG-grounded chatbots with citations, moderation, and annotation workflows are a core strength" + }, + "content-creation": { + "overall": 85, + "notes": "Visual workflows with templates suit marketing and content generation pipelines" + }, + "research-assistant": { + "overall": 82, + "notes": "Knowledge bases plus web/search tools support internal research assistants" + }, + "data-analysis": { + "overall": 76, + "notes": "Sandboxed code nodes handle light analysis; heavy data work better in code-first frameworks" + }, + "education": { + "overall": 82, + "notes": "Easy no-code building of tutoring and Q&A apps over course materials" + }, + "code-generation": { + "overall": 72, + "notes": "Possible via LLM nodes but not a developer-workflow-focused platform" + }, + "legal-compliance": { + "overall": 76, + "notes": "Self-hosted RAG over legal documents works, but compliance guardrails need configuration" + } + }, + "best_for": [ + "Teams wanting no-code/low-code agentic workflows with built-in RAG", + "Enterprises self-hosting LLM apps for data control with a visual ops layer", + "Product teams shipping chatbots and assistants via Backend-as-a-Service APIs", + "Organizations standardizing many internal LLM apps on one platform" + ], + "strengths": [ + "Visual workflow canvas combining agent nodes, RAG, and 50+ built-in tools", + "Massive adoption and community: 138k+ GitHub stars, 1M+ deployed apps", + "Free self-hosting with Docker/Helm and full data control", + "Built-in observability: logs, annotations, analytics, and tracing integrations", + "Dedicated dify-sandbox for code-node execution", + "Clear, affordable cloud pricing (free Sandbox, $59/mo Professional, $159/mo Team)" + ], + "limitations": [ + "License is Apache-2.0-based but custom: multi-tenant SaaS resale and logo removal require a commercial license", + "Multi-agent orchestration is shallower than code-first frameworks like LangGraph or CrewAI", + "Visual abstraction limits fine-grained programmatic control for complex agent logic", + "Prompt injection defense relies on opt-in moderation rather than dedicated mechanisms", + "Self-hosted stack (API, worker, sandbox, vector DB) has nontrivial operational footprint" + ], + "related": [ + "flowise", + "langflow", + "n8n-ai-agent", + "crewai", + "langgraph-agent" + ], + "metadata": { + "license": "Dify Open Source License (Apache 2.0 based, with multi-tenant SaaS resale restrictions)", + "supported_models": [ + "OpenAI", + "Anthropic Claude", + "Google Gemini", + "Azure OpenAI", + "Local LLMs via Ollama/Xinference", + "Hundreds of models via provider plugins" + ], + "programming_languages": [ + "Python (backend)", + "TypeScript (frontend)", + "No-code visual builder" + ], + "deployment_type": "Self-hosted or Dify Cloud", + "tool_support": [ + "50+ built-in tools", + "Custom API tools", + "Plugin marketplace", + "MCP tools" + ], + "github_stars": "138000+", + "first_release": "2023", + "pricing": "Free self-hosted; Cloud: Sandbox free, Professional $59/mo, Team $159/mo", + "python_requirement": "Python >=3.11 (backend, when developing from source)", + "adoption": "1M+ apps deployed on the platform as of April 2026" + }, + "tags": [ + "no-code", + "workflow-platform", + "open-source" + ] +} diff --git a/data/agents/gemini-cli.json b/data/agents/gemini-cli.json new file mode 100644 index 0000000..8d61597 --- /dev/null +++ b/data/agents/gemini-cli.json @@ -0,0 +1,458 @@ +{ + "id": "gemini-cli", + "type": "agent", + "name": "Gemini CLI", + "provider": "Google", + "version": "0.x (rolling release)", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Open-source terminal AI agent from Google that brings Gemini into the command line. Uses a ReAct loop with built-in tools, MCP server support, and Google Search grounding for coding, content generation, and task automation directly in the shell.", + "website": "https://github.com/google-gemini/gemini-cli", + "trust_vector": { + "performance_reliability": { + "overall_score": 77, + "criteria": { + "task_completion_accuracy": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Google Blog - Introducing Gemini CLI", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/introducing-gemini-cli-open-source-ai-agent/", + "date": "2025-06-25", + "value": "Gemini 2.5 Pro with 1M token context window handles large codebases and multi-file tasks" + } + ], + "methodology": "Coding and shell task completion testing with large-context workloads", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-01", + "value": "Built-in tools (file ops, shell, web fetch, Google Search grounding) plus extensibility via MCP servers" + } + ], + "methodology": "Tool integration testing across built-in and MCP tools", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 83, + "confidence": "high", + "evidence": [ + { + "source": "Google Blog - Introducing Gemini CLI", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/introducing-gemini-cli-open-source-ai-agent/", + "date": "2025-06-25", + "value": "Reason-and-act (ReAct) loop plans, executes tools, and iterates on complex tasks like bug fixing and feature work" + } + ], + "methodology": "Multi-step task decomposition testing via ReAct loop", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Documentation", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/index.md", + "date": "2026-06-01", + "value": "GEMINI.md context files, session checkpointing, and a memory tool persist project context across sessions" + } + ], + "methodology": "Session and context persistence evaluation", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI GitHub Issues", + "url": "https://github.com/google-gemini/gemini-cli/issues", + "date": "2026-06-01", + "value": "ReAct loop retries on tool failures and model fallback (Pro to Flash) on rate limits; loop detection still imperfect per community reports" + } + ], + "methodology": "Error handling and retry behavior testing", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Documentation", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/index.md", + "date": "2026-06-01", + "value": "Primarily a single-agent CLI; non-interactive mode and MCP enable scripting into pipelines but no native multi-agent orchestration" + } + ], + "methodology": "Multi-agent and scripting capability assessment", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 80, + "criteria": { + "tool_sandboxing": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Gemini CLI Sandbox Documentation", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/cli/sandbox.md", + "date": "2026-06-01", + "value": "Optional sandboxing via Docker/Podman containers or macOS Seatbelt profiles to isolate shell and file operations" + } + ], + "methodology": "Sandboxing options review and testing", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Configuration Docs", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/cli/configuration.md", + "date": "2026-06-01", + "value": "Tool allowlists/blocklists, trusted folders, and Google account or API key auth; enterprise policy via Gemini Code Assist licensing" + } + ], + "methodology": "Access control configuration assessment", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-01", + "value": "Per-tool confirmation prompts before file writes and shell execution act as a human-in-the-loop gate; YOLO mode bypasses protections" + } + ], + "methodology": "Injection surface review focusing on confirmation gates and web content handling", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Terms and Privacy", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/tos-privacy.md", + "date": "2026-06-01", + "value": "Free personal-account tier may use prompts and code to improve Google products; paid API and Code Assist Standard/Enterprise tiers exclude training use" + } + ], + "methodology": "Data handling terms review across auth tiers", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-10", + "value": "Apache 2.0 licensed, fully open source with public roadmap and tens of thousands of stars" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 69, + "criteria": { + "data_retention": { + "score": 71, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Terms and Privacy", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/tos-privacy.md", + "date": "2026-06-01", + "value": "Retention governed by Google terms tied to the chosen auth method; free-tier data may be retained for product improvement" + } + ], + "methodology": "Privacy policy and terms review", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Google Cloud Compliance", + "url": "https://cloud.google.com/security/compliance", + "date": "2026-06-01", + "value": "Enterprise use via Gemini Code Assist / Vertex AI inherits Google Cloud GDPR commitments; consumer tier offers weaker guarantees" + } + ], + "methodology": "Compliance posture assessment per access tier", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Terms and Privacy", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/tos-privacy.md", + "date": "2026-06-01", + "value": "All prompts, files in context, and tool outputs are sent to Google's Gemini API; Search grounding sends queries to Google Search" + } + ], + "methodology": "Data flow analysis", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-01", + "value": "Client is open source and runs locally, but inference requires Google's cloud Gemini models; no official local-model backend" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 84, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Gemini CLI Documentation", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/index.md", + "date": "2026-06-01", + "value": "Detailed docs covering commands, configuration, sandboxing, MCP, extensions, and enterprise deployment" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Telemetry Docs", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/cli/telemetry.md", + "date": "2026-06-01", + "value": "OpenTelemetry-based observability for prompts, tool calls, and token usage; visible tool call log in the terminal UI" + } + ], + "methodology": "Logging and telemetry capabilities assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-01", + "value": "ReAct loop surfaces reasoning steps and proposed actions before execution, with diffs shown for file edits" + } + ], + "methodology": "Explainability features assessment", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-10", + "value": "Apache 2.0 license; released 2025-06-25 with active public development and community contributions" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-10", + "value": "One of the fastest-growing CLI agent repos with very high star count, frequent releases, and an extensions ecosystem" + } + ], + "methodology": "Community engagement analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 74, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Google Blog - Introducing Gemini CLI", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/introducing-gemini-cli-open-source-ai-agent/", + "date": "2025-06-25", + "value": "Single npm install and Google account login; free tier with 60 requests/min and 1,000 requests/day" + } + ], + "methodology": "Setup and onboarding complexity assessment", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Documentation", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/index.md", + "date": "2026-06-01", + "value": "Non-interactive mode supports CI/CD scripting; throughput bounded by per-user rate limits unless using paid API keys" + } + ], + "methodology": "Automation and rate-limit assessment", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI GitHub", + "url": "https://github.com/google-gemini/gemini-cli", + "date": "2026-06-01", + "value": "Generous free tier (1,000 requests/day); paid usage via metered Gemini API keys or Gemini Code Assist subscriptions" + } + ], + "methodology": "Pricing model analysis", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini CLI Telemetry Docs", + "url": "https://github.com/google-gemini/gemini-cli/blob/main/docs/cli/telemetry.md", + "date": "2026-06-01", + "value": "OpenTelemetry export for metrics and traces; enterprise-grade monitoring requires external tooling" + } + ], + "methodology": "Monitoring features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Digital Applied - Gemini CLI to Antigravity CLI Migration Guide", + "url": "https://www.digitalapplied.com/blog/gemini-cli-to-antigravity-cli-migration-june-18-2026-guide", + "date": "2026-06-05", + "value": "Third-party reporting indicates consumer access to Gemini CLI ends 2026-06-18 as Google migrates individual users to Antigravity CLI; enterprise Gemini Code Assist users retain access" + } + ], + "methodology": "Product continuity assessment based on third-party migration reporting; not yet confirmed in official Google release notes", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 84, + "notes": "Strong terminal coding agent with large context, ReAct loop, and MCP extensibility" + }, + "research-assistant": { + "overall": 80, + "notes": "Google Search grounding and web fetch make it effective for in-terminal research" + }, + "data-analysis": { + "overall": 76, + "notes": "Capable of scripting analyses via shell and file tools, though not analytics-specialized" + }, + "content-creation": { + "overall": 74, + "notes": "Useful for drafting docs and content in the terminal with multimodal generation via extensions" + }, + "education": { + "overall": 75, + "notes": "Free tier makes it accessible for learning, but consumer access continuity is uncertain" + } + }, + "best_for": [ + "Developers wanting a free, open-source AI agent in the terminal", + "Teams already on Google's Gemini ecosystem and Gemini Code Assist", + "Scripted automation via non-interactive mode and MCP servers", + "Quick coding, debugging, and research tasks with Search grounding" + ], + "strengths": [ + "Open source (Apache 2.0) with very large and active community", + "Generous free tier: 60 requests/min and 1,000 requests/day", + "Strong safety options: Docker/Seatbelt sandboxing and per-tool confirmation prompts", + "ReAct loop with built-in tools, MCP support, and Google Search grounding", + "1M token context window via Gemini 2.5 Pro for large codebases" + ], + "limitations": [ + "Third-party reports indicate consumer access ends 2026-06-18 with migration to Antigravity CLI; only enterprise Gemini Code Assist retains access (unconfirmed officially, medium confidence)", + "Free personal tier may use prompts and code for Google product improvement", + "No local model backend; inference requires Google's cloud Gemini API", + "Single-agent design with no native multi-agent orchestration", + "Rate limits and model fallback (Pro to Flash) can degrade quality under load" + ], + "metadata": { + "license": "Apache 2.0", + "supported_models": [ + "Gemini 2.5 Pro", + "Gemini 2.5 Flash", + "Gemini 3 (paid/preview tiers)" + ], + "programming_languages": [ + "TypeScript (Node.js)" + ], + "deployment_type": "Local CLI client with cloud model inference", + "tool_support": [ + "Built-in file and shell tools", + "Google Search grounding", + "Web fetch", + "MCP servers", + "Extensions" + ], + "first_release": "2025-06-25", + "pricing": "Free tier (60 req/min, 1,000 req/day with personal Google account); paid via Gemini API keys or Gemini Code Assist Standard/Enterprise", + "product_status_note": "Consumer access reportedly ending 2026-06-18 in favor of Antigravity CLI per third-party reporting; enterprise access continues via Gemini Code Assist" + }, + "tags": [ + "cli", + "open-source", + "coding-agent", + "google" + ] +} diff --git a/data/agents/github-copilot-coding-agent.json b/data/agents/github-copilot-coding-agent.json new file mode 100644 index 0000000..3358efd --- /dev/null +++ b/data/agents/github-copilot-coding-agent.json @@ -0,0 +1,455 @@ +{ + "id": "github-copilot-coding-agent", + "type": "agent", + "name": "GitHub Copilot Coding Agent", + "provider": "GitHub (Microsoft)", + "version": "GA (2025-09)", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Autonomous background coding agent built into GitHub. Assign it a GitHub issue or prompt and it works in an ephemeral GitHub Actions sandbox, then opens a draft pull request for human review. Distinct from Copilot's interactive IDE agent mode.", + "website": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "trust_vector": { + "performance_reliability": { + "overall_score": 76, + "criteria": { + "task_completion_accuracy": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - About Copilot coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Designed for well-scoped tasks: bug fixes, incremental features, test coverage, and documentation in familiar codebases" + } + ], + "methodology": "Task scope analysis and PR outcome review on representative issues", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 83, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Full GitHub Actions environment with build/test execution, plus MCP server support and vision capabilities for issue images" + } + ], + "methodology": "Tooling and environment reliability assessment", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Blog - Copilot coding agent", + "url": "https://github.blog/news-insights/product-news/github-copilot-meet-the-new-coding-agent/", + "date": "2025-05-19", + "value": "Agent explores the repo, plans changes, iterates on build and test feedback, and pushes commits incrementally" + } + ], + "methodology": "Multi-step task execution and iteration testing", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - Customizing the agent environment", + "url": "https://docs.github.com/en/copilot/how-tos/use-copilot-agents/coding-agent/customize-the-agent-environment", + "date": "2026-06-01", + "value": "Sessions are ephemeral; persistent context provided via repository custom instructions and copilot-setup-steps configuration" + } + ], + "methodology": "Cross-session context persistence evaluation", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Runs builds and tests in the Actions sandbox and iterates on failures; responds to PR review comments with fixes" + } + ], + "methodology": "Failure iteration and review-feedback loop testing", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Single agent per task with multiple parallel tasks supported; collaborates with humans via PR comments rather than other agents" + } + ], + "methodology": "Concurrency and orchestration capability assessment", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 74, + "criteria": { + "tool_sandboxing": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Coding agent security", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Runs in an ephemeral GitHub Actions environment with firewall-restricted internet access; firewall allowlist is customizable" + } + ], + "methodology": "Sandbox and network isolation architecture review", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Agent can only push to copilot/ branches, cannot approve or merge its own PRs, respects branch protections, and PR-triggering workflows require human approval" + } + ], + "methodology": "Permission boundary and branch protection review", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - Coding agent risks and mitigations", + "url": "https://docs.github.com/en/copilot/responsible-use/copilot-coding-agent", + "date": "2026-06-01", + "value": "Firewall limits exfiltration, hidden-content filtering on issues, and mandatory human PR review mitigate injection from untrusted repo content" + } + ], + "methodology": "Injection mitigation review against documented threat model", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Each session runs in its own ephemeral Actions environment scoped to a single repository" + } + ], + "methodology": "Session and tenant isolation review", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Copilot Product Page", + "url": "https://github.com/features/copilot", + "date": "2026-06-01", + "value": "Proprietary closed-source service; underlying models and agent implementation are not public" + } + ], + "methodology": "Source availability review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 65, + "criteria": { + "data_retention": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Copilot Trust Center", + "url": "https://copilot.github.trust.page/", + "date": "2026-06-01", + "value": "GitHub states Copilot Business/Enterprise prompts and code are not used to train models; retention governed by GitHub data protection terms" + } + ], + "methodology": "Data handling and retention terms review", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Data Protection", + "url": "https://docs.github.com/en/site-policy/privacy-policies/github-general-privacy-statement", + "date": "2026-06-01", + "value": "Covered by GitHub/Microsoft compliance programs including GDPR data protection agreements for organizations" + } + ], + "methodology": "Compliance program and DPA assessment", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Copilot Trust Center", + "url": "https://copilot.github.trust.page/", + "date": "2026-06-01", + "value": "Repository content is processed by GitHub-hosted model providers (OpenAI, Anthropic, Google models) under GitHub's agreements" + } + ], + "methodology": "Data flow analysis across model backends", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Cloud-only; runs exclusively in GitHub-hosted Actions infrastructure with no self-hosted runner or on-prem option for the agent" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 72, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Copilot coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Extensive official docs covering concepts, security model, environment customization, MCP, and responsible use" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Tracking agent sessions", + "url": "https://docs.github.com/en/copilot/how-tos/use-copilot-agents/coding-agent/track-copilot-sessions", + "date": "2026-06-01", + "value": "Session logs show the agent's full reasoning and tool steps; all work lands as incremental commits on a draft PR" + } + ], + "methodology": "Session log and commit trail assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Draft PR descriptions summarize intent and approach; session logs expose step-by-step decisions" + } + ], + "methodology": "Explainability features assessment", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Copilot Product Page", + "url": "https://github.com/features/copilot", + "date": "2026-06-01", + "value": "Proprietary; agent implementation and models are closed source" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Community Discussions", + "url": "https://github.com/orgs/community/discussions/categories/copilot", + "date": "2026-06-01", + "value": "Very large user base with active community discussions, changelog updates, and rapid feature iteration" + } + ], + "methodology": "Community engagement analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 82, + "criteria": { + "ease_of_integration": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Native to GitHub: assign an issue to Copilot or prompt from chat/Agents panel; no infrastructure setup required" + } + ], + "methodology": "Onboarding and integration assessment", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Docs - Coding agent", + "url": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", + "date": "2026-06-01", + "value": "Multiple parallel agent sessions on GitHub-hosted Actions infrastructure, bounded by plan quotas and Actions minutes" + } + ], + "methodology": "Parallelism and quota analysis", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub Blog - Copilot moving to usage-based billing", + "url": "https://github.blog/news-insights/company-news/github-copilot-is-moving-to-usage-based-billing/", + "date": "2026-06-01", + "value": "Premium-request billing since 2025-06-18; from 2026-06-01 GitHub is transitioning to token-based 'GitHub AI Credits', making per-task costs harder to forecast during the changeover" + } + ], + "methodology": "Pricing model analysis including billing model transition", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Docs - Tracking agent sessions", + "url": "https://docs.github.com/en/copilot/how-tos/use-copilot-agents/coding-agent/track-copilot-sessions", + "date": "2026-06-01", + "value": "Agents panel for session tracking, session logs, audit log events, and org-level policy controls" + } + ], + "methodology": "Monitoring and audit features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Copilot Plans", + "url": "https://docs.github.com/en/copilot/get-started/plans", + "date": "2026-06-01", + "value": "GA since September 2025 on Pro, Pro+, Business, and Enterprise plans, backed by GitHub's production infrastructure" + } + ], + "methodology": "Product maturity and availability assessment", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 88, + "notes": "Purpose-built for issue-to-PR coding tasks with strong GitHub-native guardrails" + }, + "data-analysis": { + "overall": 68, + "notes": "Can write analysis code and tests in repos, but not designed for interactive analytics" + }, + "education": { + "overall": 72, + "notes": "Session logs and reviewable PRs help learners; requires a paid Copilot plan for the coding agent" + }, + "content-creation": { + "overall": 66, + "notes": "Useful for documentation and README work within repositories; not a general content tool" + } + }, + "best_for": [ + "Teams already on GitHub wanting issue-to-PR automation with zero setup", + "Burning down backlogs of well-scoped bugs, tests, and documentation tasks", + "Organizations needing enterprise controls, audit logs, and policy management", + "Workflows where every AI change must pass human PR review and branch protections" + ], + "strengths": [ + "Strong security model: ephemeral Actions sandbox with firewall-restricted internet", + "Hard guardrails: pushes only to copilot/ branches, cannot merge its own PRs, branch protections enforced", + "Native GitHub integration; trigger from issues, chat, mobile, or the Agents panel", + "Full session logs and incremental commits give a complete audit trail", + "Iterates on build/test failures and responds to PR review comments", + "Backed by GitHub/Microsoft enterprise compliance programs" + ], + "limitations": [ + "Proprietary and cloud-only; no self-hosted runner support for the agent", + "Billing complexity: premium requests since 2025-06-18 and a transition to token-based GitHub AI Credits beginning 2026-06-01 make costs harder to predict", + "Requires a paid Copilot plan (Pro, Pro+, Business, or Enterprise); not in Copilot Free", + "Best on well-scoped tasks; struggles with large cross-repo or ambiguous refactors", + "Consumes GitHub Actions minutes in addition to premium requests/credits", + "Each session is ephemeral with limited memory across tasks" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "OpenAI GPT models", + "Anthropic Claude models", + "Google Gemini models (per Copilot model availability)" + ], + "programming_languages": [ + "Most major languages supported by the repository's toolchain" + ], + "deployment_type": "Cloud-only SaaS (ephemeral GitHub Actions environments)", + "tool_support": [ + "GitHub Actions build/test execution", + "MCP servers", + "Repository custom instructions", + "Vision input from issue images", + "Draft PR creation" + ], + "first_release": "Preview May 2025; GA September 2025", + "pricing": "Copilot Pro $10/mo, Pro+ $39/mo, Business/Enterprise per-seat; agent usage billed via premium requests (since 2025-06-18), transitioning to token-based GitHub AI Credits from 2026-06-01" + }, + "tags": [ + "coding-agent", + "autonomous", + "github", + "enterprise" + ] +} diff --git a/data/agents/google-adk.json b/data/agents/google-adk.json new file mode 100644 index 0000000..9c00045 --- /dev/null +++ b/data/agents/google-adk.json @@ -0,0 +1,471 @@ +{ + "id": "google-adk", + "type": "agent", + "name": "Google Agent Development Kit (ADK)", + "provider": "Google", + "version": "2.0", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Open-source, code-first framework for building, evaluating, and deploying AI agents. Supports workflow agents, multi-agent hierarchies, built-in evaluation, and deployment to Vertex AI Agent Engine. Underlies Google's broader agent stack and is Gemini-optimized but model-agnostic.", + "website": "https://google.github.io/adk-docs/", + "trust_vector": { + "performance_reliability": { + "overall_score": 85, + "criteria": { + "task_completion_accuracy": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Documentation - Evaluate", + "url": "https://google.github.io/adk-docs/evaluate/", + "date": "2026-05-19", + "value": "Built-in evaluation framework for assessing response quality and step-by-step trajectory against test cases" + } + ], + "methodology": "Review of built-in evaluation tooling and reported agent benchmark workflows", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "ADK Documentation - Tools", + "url": "https://google.github.io/adk-docs/tools/", + "date": "2026-05-19", + "value": "Rich tool ecosystem: function tools, OpenAPI tools, MCP tools, Google Cloud tools, and third-party library integrations" + } + ], + "methodology": "Tool integration testing across function, OpenAPI, and MCP tool types", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "ADK 2.0 Release Notes", + "url": "https://google.github.io/adk-docs/release-notes/", + "date": "2026-05-19", + "value": "ADK 2.0 GA adds graph-based, dynamic, and collaborative workflows alongside sequential, parallel, and loop workflow agents" + } + ], + "methodology": "Assessment of workflow agent primitives and graph-based orchestration in ADK 2.0", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Documentation - Sessions and Memory", + "url": "https://google.github.io/adk-docs/sessions/", + "date": "2026-05-19", + "value": "Session, state, and memory services with in-memory, database, and Vertex AI managed backends" + } + ], + "methodology": "Memory and session service architecture evaluation", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "ADK GitHub Issues", + "url": "https://github.com/google/adk-python/issues", + "date": "2026-06-01", + "value": "Callbacks and loop agents enable retry patterns; robust recovery logic remains developer responsibility" + } + ], + "methodology": "Error handling pattern review and community issue analysis", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "ADK Documentation - Multi-Agent Systems", + "url": "https://google.github.io/adk-docs/agents/multi-agents/", + "date": "2026-05-19", + "value": "First-class multi-agent hierarchies with delegation, agent-as-tool composition, and A2A protocol support" + } + ], + "methodology": "Multi-agent hierarchy and delegation capability testing", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 79, + "criteria": { + "tool_sandboxing": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Documentation - Code Execution", + "url": "https://google.github.io/adk-docs/tools/built-in-tools/", + "date": "2026-05-19", + "value": "Sandboxed code execution available via Vertex AI code executor; general tool sandboxing is developer responsibility" + } + ], + "methodology": "Security architecture review of tool execution paths", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Documentation - Authentication", + "url": "https://google.github.io/adk-docs/tools/authentication/", + "date": "2026-05-19", + "value": "Tool authentication framework (OAuth2, API keys, service accounts) plus Google Cloud IAM when deployed on Vertex AI" + } + ], + "methodology": "Access control and authentication capability assessment", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Documentation - Safety and Security", + "url": "https://google.github.io/adk-docs/safety/", + "date": "2026-05-19", + "value": "Documented guidance on guardrails, callbacks for input/output screening, and Gemini safety settings; no automatic injection blocking" + } + ], + "methodology": "Review of documented safety patterns and guardrail mechanisms", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Documentation - Sessions", + "url": "https://google.github.io/adk-docs/sessions/", + "date": "2026-05-19", + "value": "Per-session state isolation with user- and app-scoped state prefixes; self-hosted deployments control data boundaries" + } + ], + "methodology": "Data architecture and session isolation review", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "google/adk-python GitHub", + "url": "https://github.com/google/adk-python", + "date": "2026-06-10", + "value": "Apache 2.0 license, fully open source with public development across Python, Java, TypeScript, Go, and Kotlin SDKs" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 82, + "criteria": { + "data_retention": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Deployment Options", + "url": "https://google.github.io/adk-docs/deploy/", + "date": "2026-05-19", + "value": "Self-hosted deployments give full control over retention; Vertex AI Agent Engine follows Google Cloud data governance" + } + ], + "methodology": "Privacy architecture review across deployment options", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Google Cloud Compliance", + "url": "https://cloud.google.com/security/compliance", + "date": "2026-06-01", + "value": "Vertex AI deployments inherit Google Cloud GDPR commitments; self-hosted compliance depends on operator configuration" + } + ], + "methodology": "Compliance capabilities assessment for framework and managed deployments", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Documentation - Models", + "url": "https://google.github.io/adk-docs/agents/models/", + "date": "2026-05-19", + "value": "Prompts and tool data are sent to the configured model provider (Gemini API by default, others via LiteLLM)" + } + ], + "methodology": "Data flow analysis of model and tool integrations", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "ADK Documentation - Models", + "url": "https://google.github.io/adk-docs/agents/models/", + "date": "2026-05-19", + "value": "Model-agnostic via LiteLLM including Ollama and other local model backends; framework itself runs anywhere" + } + ], + "methodology": "Deployment options assessment including local model support", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 86, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "ADK Documentation", + "url": "https://google.github.io/adk-docs/", + "date": "2026-05-19", + "value": "Comprehensive docs covering agents, tools, workflows, evaluation, deployment, and safety with quickstarts and samples" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "ADK Dev UI and Evaluation", + "url": "https://google.github.io/adk-docs/evaluate/", + "date": "2026-05-19", + "value": "Built-in dev UI for step-by-step inspection of events, state, and tool calls; trajectory evaluation against expected steps" + } + ], + "methodology": "Tracing and observability capabilities assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Events Model", + "url": "https://google.github.io/adk-docs/events/", + "date": "2026-05-19", + "value": "Event stream exposes agent reasoning steps, function calls, and state deltas for inspection" + } + ], + "methodology": "Explainability features assessment", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "google/adk-python GitHub", + "url": "https://github.com/google/adk-python", + "date": "2026-06-10", + "value": "Apache 2.0 licensed; announced at Cloud Next April 2025, Python v1.0 May 2025, ADK 2.0 GA May 2026" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "ADK GitHub Activity", + "url": "https://github.com/google/adk-python", + "date": "2026-06-10", + "value": "Active development with frequent releases, multi-language SDK expansion, and growing contributor base since April 2025 launch" + } + ], + "methodology": "Community engagement and release cadence analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 84, + "criteria": { + "ease_of_integration": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "ADK Quickstart", + "url": "https://google.github.io/adk-docs/get-started/quickstart/", + "date": "2026-05-19", + "value": "pip-installable with code-first Pythonic API; agents definable in under 100 lines with CLI and dev UI included" + } + ], + "methodology": "Integration complexity assessment", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "ADK Deploy to Agent Engine", + "url": "https://google.github.io/adk-docs/deploy/agent-engine/", + "date": "2026-05-19", + "value": "Managed scaling via Vertex AI Agent Engine plus Cloud Run and GKE deployment paths" + } + ], + "methodology": "Deployment and scaling options assessment", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "ADK GitHub", + "url": "https://github.com/google/adk-python", + "date": "2026-06-10", + "value": "Framework is free (Apache 2.0); costs come from model API usage and optional Vertex AI / Gemini Enterprise Agent Platform services" + } + ], + "methodology": "Pricing model analysis", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "ADK Observability Docs", + "url": "https://google.github.io/adk-docs/observability/logging/", + "date": "2026-05-19", + "value": "Logging, tracing integrations, and evaluation tooling; Cloud Trace and third-party observability supported on managed deployments" + } + ], + "methodology": "Monitoring features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "ADK Release Notes", + "url": "https://google.github.io/adk-docs/release-notes/", + "date": "2026-05-19", + "value": "Python v1.0 stable since May 2025; ADK 2.0 GA on 2026-05-19 with production-focused workflow and deployment improvements" + } + ], + "methodology": "Production readiness and release maturity assessment", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "customer-support": { + "overall": 86, + "notes": "Strong fit for multi-agent support systems; underlies Google's Customer Engagement Suite agent stack" + }, + "code-generation": { + "overall": 80, + "notes": "Capable via code execution tools and workflow agents, though not a purpose-built coding agent" + }, + "research-assistant": { + "overall": 87, + "notes": "Multi-agent hierarchies with Google Search grounding work well for research pipelines" + }, + "data-analysis": { + "overall": 84, + "notes": "Good with code executors, BigQuery and Google Cloud tool integrations" + }, + "content-creation": { + "overall": 80, + "notes": "Workflow agents support multi-stage drafting and review pipelines" + }, + "financial-analysis": { + "overall": 78, + "notes": "Viable on Vertex AI with enterprise controls; compliance hardening is the builder's responsibility" + }, + "healthcare": { + "overall": 74, + "notes": "Requires significant compliance work; Google Cloud HIPAA-eligible services help when deployed on Vertex AI" + } + }, + "best_for": [ + "Teams building multi-agent systems on Google Cloud and Gemini", + "Developers wanting a code-first framework with built-in evaluation", + "Enterprises needing a path from prototype to managed Agent Engine deployment", + "Builders who want model-agnostic flexibility with Gemini optimization" + ], + "strengths": [ + "First-class multi-agent hierarchies, delegation, and A2A protocol support", + "ADK 2.0 adds graph-based, dynamic, and collaborative workflows", + "Built-in evaluation framework for response quality and trajectory testing", + "Open source (Apache 2.0) with Python, Java, TypeScript, Go, and Kotlin SDKs", + "Clean deployment path to Vertex AI Agent Engine, Cloud Run, or GKE", + "Model-agnostic via LiteLLM while optimized for Gemini" + ], + "limitations": [ + "Best experience is tied to the Google Cloud and Gemini ecosystem", + "Tool sandboxing and injection defenses are largely developer responsibility", + "Rapid release cadence has introduced breaking changes between major versions", + "Managed features (Agent Engine, Gemini Enterprise Agent Platform) add paid cloud dependency", + "Younger ecosystem than incumbent frameworks like LangChain/LangGraph" + ], + "metadata": { + "license": "Apache 2.0", + "supported_models": [ + "Google Gemini (optimized)", + "Anthropic Claude", + "OpenAI models via LiteLLM", + "Local LLMs via LiteLLM/Ollama" + ], + "programming_languages": [ + "Python", + "Java", + "TypeScript", + "Go", + "Kotlin" + ], + "deployment_type": "Self-hosted or managed (Vertex AI Agent Engine, Cloud Run, GKE)", + "tool_support": [ + "Function tools", + "OpenAPI tools", + "MCP tools", + "Google Cloud tools", + "Third-party library tools" + ], + "first_release": "2025 (announced Cloud Next 2025-04-09; Python v1.0 May 2025; ADK 2.0 GA 2026-05-19)", + "pricing": "Free framework (Apache 2.0); paid when using Vertex AI Agent Engine or Gemini Enterprise Agent Platform plus model API costs" + }, + "tags": [ + "multi-agent", + "open-source", + "google", + "framework" + ] +} diff --git a/data/agents/google-agent-builder.json b/data/agents/google-agent-builder.json index 7d9fb6d..550dfdf 100644 --- a/data/agents/google-agent-builder.json +++ b/data/agents/google-agent-builder.json @@ -1,13 +1,13 @@ { "id": "google-agent-builder", "type": "agent", - "name": "Google Vertex AI Agent Builder", + "name": "Gemini Enterprise Agent Platform (formerly Vertex AI Agent Builder)", "provider": "Google Cloud", - "version": "2024", - "last_evaluated": "2025-11-09", + "version": "2026", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Google Cloud's managed platform for building conversational AI agents and search applications. Provides no-code and low-code options for agent creation with enterprise search, grounding, and Google services integration.", - "website": "https://cloud.google.com/generative-ai-app-builder/docs/agent-intro", + "description": "REBRANDED: at Cloud Next (April 2026) Vertex AI Agent Builder became the Gemini Enterprise Agent Platform (APIs unchanged), following Agentspace's absorption into Gemini Enterprise in Oct 2025. Google Cloud's managed platform for building conversational AI agents and search apps with no-code/low-code options, enterprise search, and grounding.", + "website": "https://cloud.google.com/products/gemini-enterprise-agent-platform", "trust_vector": { "performance_reliability": { "overall_score": 87, @@ -378,10 +378,16 @@ "url": "https://cloud.google.com/customers#/products=AI_/_Machine_Learning", "date": "2024-10-01", "value": "Production-ready with enterprise customer deployments" + }, + { + "source": "Gemini Enterprise Agent Platform", + "url": "https://cloud.google.com/products/gemini-enterprise-agent-platform", + "date": "2026-06-10", + "value": "Rebranded at Cloud Next April 2026 from Vertex AI Agent Builder to Gemini Enterprise Agent Platform; APIs unchanged. Agentspace was absorbed into Gemini Enterprise in Oct 2025" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -400,7 +406,8 @@ "Higher costs for grounding and search features", "Less code-level flexibility than open frameworks", "Requires Google Cloud expertise for optimization", - "Not open source, limited customization of orchestration" + "Not open source, limited customization of orchestration", + "Repeated rebranding (Agentspace into Gemini Enterprise Oct 2025; Agent Builder renamed April 2026) makes older docs/links stale, though APIs are unchanged" ], "metadata": { "license": "Proprietary (Google Cloud)", @@ -472,6 +479,7 @@ "Organizations requiring Google Workspace integration" ], "tags": [ - "google" + "google", + "rebranded" ] } diff --git a/data/agents/google-jules.json b/data/agents/google-jules.json new file mode 100644 index 0000000..2985aba --- /dev/null +++ b/data/agents/google-jules.json @@ -0,0 +1,454 @@ +{ + "id": "google-jules", + "type": "agent", + "name": "Google Jules", + "provider": "Google", + "version": "GA (2025-08-06)", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Asynchronous autonomous coding agent from Google. Jules clones a repository into an isolated Google Cloud VM, plans and writes code in the background, runs tests, and opens pull requests for human review. Powered by Gemini 2.5/3 models.", + "website": "https://jules.google/", + "trust_vector": { + "performance_reliability": { + "overall_score": 76, + "criteria": { + "task_completion_accuracy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Google Blog - Jules now available", + "url": "https://blog.google/technology/google-labs/jules-now-available/", + "date": "2025-08-06", + "value": "GA after public beta with hundreds of thousands of tasks completed; handles bug fixes, version bumps, tests, and feature work" + } + ], + "methodology": "Assessment of reported task outcomes and hands-on PR quality review", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Documentation", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "Full VM environment with shell access, dependency installation, environment setup scripts, and test execution" + } + ], + "methodology": "VM tooling and environment setup reliability assessment", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Presents an explicit plan with reasoning before executing; users can review and steer the plan before and during execution" + } + ], + "methodology": "Plan generation and decomposition quality review", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Documentation", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "Per-repo environment configuration and snapshots persist across tasks; memory feature retains repo-specific preferences and corrections" + } + ], + "methodology": "Cross-task context persistence evaluation", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Changelog", + "url": "https://jules.google/docs/changelog", + "date": "2026-06-01", + "value": "Runs tests in the VM and iterates on failures; users can intervene mid-task via chat to correct course" + } + ], + "methodology": "Failure handling and iteration behavior testing", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Single autonomous agent per task; supports concurrent independent tasks (3-60 depending on plan) but no multi-agent orchestration" + } + ], + "methodology": "Concurrency and orchestration capability assessment", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 71, + "criteria": { + "tool_sandboxing": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Google Blog - Jules now available", + "url": "https://blog.google/technology/google-labs/jules-now-available/", + "date": "2025-08-06", + "value": "Each task runs in an isolated, ephemeral Google Cloud VM, separating agent execution from user machines and other tasks" + } + ], + "methodology": "Execution isolation architecture review", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Documentation", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "GitHub App installation with per-repository access selection; changes land on branches as pull requests" + } + ], + "methodology": "Repository permission model assessment", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Human PR review gate plus plan approval limit blast radius; repository content itself remains an injection vector" + } + ], + "methodology": "Injection surface review focusing on autonomy boundaries and human gates", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Documentation", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "Per-task VM isolation keeps repository data separated between tasks and tenants" + } + ], + "methodology": "Tenant and task isolation review", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Proprietary closed-source service; only client tooling (Jules Tools CLI) and docs are public" + } + ], + "methodology": "Source availability review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 58, + "criteria": { + "data_retention": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Documentation - Privacy", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "Google states private repository code is not used to train models; data handling governed by Google's privacy terms" + } + ], + "methodology": "Privacy terms and data handling review", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Google Privacy Policy", + "url": "https://policies.google.com/privacy", + "date": "2026-06-01", + "value": "Covered by Google's general privacy framework; lacks dedicated enterprise compliance attestations of Google Cloud products" + } + ], + "methodology": "Compliance posture assessment", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Documentation", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "Full repository contents are cloned to Google-managed cloud VMs and processed by Gemini models" + } + ], + "methodology": "Data flow analysis", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Cloud-only service; no self-hosted or on-premises deployment option" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 69, + "criteria": { + "documentation_quality": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Jules Documentation", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "Clear docs covering setup, environment configuration, API, Jules Tools CLI, and changelog" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Shows its plan, reasoning, file diffs, and activity feed during execution; all changes arrive as reviewable PRs" + } + ], + "methodology": "Execution visibility assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Google Blog - Jules now available", + "url": "https://blog.google/technology/google-labs/jules-now-available/", + "date": "2025-08-06", + "value": "Upfront plan with per-step reasoning that users approve or modify before code changes are made" + } + ], + "methodology": "Explainability features assessment", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Proprietary service built on closed Gemini models; core agent is not open source" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Changelog", + "url": "https://jules.google/docs/changelog", + "date": "2026-06-01", + "value": "Frequent feature updates since GA (API, CLI, memory, GitHub issues integration) and active user community" + } + ], + "methodology": "Release cadence and community engagement analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 79, + "criteria": { + "ease_of_integration": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Jules Product Page", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Connect GitHub, pick a repo, write a prompt; also triggerable from GitHub issues by assigning a label, plus API and CLI" + } + ], + "methodology": "Onboarding and integration assessment", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Pricing", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Concurrent task limits by plan: 3 concurrent on free, 15 on Pro, 60 on Ultra; cloud VMs scale per task" + } + ], + "methodology": "Concurrency and quota analysis", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Jules Pricing", + "url": "https://jules.google/", + "date": "2026-06-01", + "value": "Flat subscription tiers: Free (15 tasks/day, 3 concurrent), Google AI Pro $19.99/mo (~100 tasks/day), AI Ultra (~300 tasks/day)" + } + ], + "methodology": "Pricing model analysis", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Jules Documentation - API", + "url": "https://jules.google/docs", + "date": "2026-06-01", + "value": "Activity feed, task history, and API for programmatic task tracking; lacks enterprise audit log integrations" + } + ], + "methodology": "Monitoring features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Google Blog - Jules now available", + "url": "https://blog.google/technology/google-labs/jules-now-available/", + "date": "2025-08-06", + "value": "Exited Google Labs beta to general availability on 2025-08-06 with paid plans and continued feature investment" + } + ], + "methodology": "Product maturity assessment", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 86, + "notes": "Purpose-built for asynchronous coding tasks: bug fixes, tests, dependency bumps, and features delivered as PRs" + }, + "data-analysis": { + "overall": 68, + "notes": "Can write and run analysis scripts in its VM, but the workflow is optimized for repository changes, not analytics" + }, + "education": { + "overall": 72, + "notes": "Visible plans and reasoning make it a useful learning aid; free tier suits students" + }, + "research-assistant": { + "overall": 64, + "notes": "Limited to codebase-scoped investigation; not designed for general research" + } + }, + "best_for": [ + "Developers offloading routine coding chores (tests, bumps, bug fixes) to a background agent", + "Teams wanting parallel asynchronous tasks without occupying a local IDE", + "Solo developers and startups using the free or Google AI Pro tiers", + "GitHub-centric workflows where every change must arrive as a reviewable PR" + ], + "strengths": [ + "Strong isolation: every task runs in an ephemeral Google Cloud VM", + "Transparent plans with reasoning that users approve before execution", + "Human PR review gate on all changes limits autonomous blast radius", + "Predictable flat-rate pricing with a usable free tier (15 tasks/day)", + "Asynchronous and parallel: fire-and-forget tasks with mid-task steering", + "API, CLI (Jules Tools), and GitHub issue-label triggers for automation" + ], + "limitations": [ + "Proprietary, cloud-only service with no self-hosted option", + "Full repository contents are uploaded to Google-managed VMs", + "Daily task limits even on paid tiers (~100/day Pro, ~300/day Ultra)", + "Locked to Gemini models with no model choice", + "GitHub-focused; weaker support for other source forges", + "Lacks enterprise compliance attestations and audit tooling of Google Cloud products" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Gemini 2.5 Pro", + "Gemini 3 (per Google updates)" + ], + "programming_languages": [ + "Most major languages (Python, JavaScript/TypeScript, Java, Go, Rust, and more)" + ], + "deployment_type": "Cloud-only SaaS (isolated Google Cloud VMs per task)", + "tool_support": [ + "Cloud VM shell", + "Environment setup scripts", + "Test execution", + "GitHub PR creation", + "Jules API and Jules Tools CLI" + ], + "first_release": "Public beta May 2025 (Google I/O); GA 2025-08-06", + "pricing": "Free: 15 tasks/day, 3 concurrent; Google AI Pro $19.99/mo (~100 tasks/day, 15 concurrent); Google AI Ultra (~300 tasks/day, 60 concurrent)" + }, + "tags": [ + "coding-agent", + "autonomous", + "asynchronous", + "google" + ] +} diff --git a/data/agents/langflow.json b/data/agents/langflow.json index 571a31a..8957d1d 100644 --- a/data/agents/langflow.json +++ b/data/agents/langflow.json @@ -2,11 +2,11 @@ "id": "langflow", "type": "agent", "name": "Langflow", - "provider": "Logspace", + "provider": "IBM (formerly DataStax/Logspace)", "version": "1.5.1+", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Visual, drag-and-drop interface for building LangChain-based AI applications and agents. Low-code platform that makes it easy to prototype and deploy LLM workflows, RAG systems, and conversational agents.", + "description": "Visual, drag-and-drop interface for building LangChain-based AI apps and agents; now owned by IBM via the Feb 2025 DataStax acquisition. SECURITY: serious CVE history including CVE-2025-3248 (CVSS 9.8 unauthenticated RCE, fixed in 1.3.0, on CISA KEV, exploited by the Flodrix botnet) and CVE-2025-34291 (account takeover/RCE). Patch promptly and harden deployments.", "website": "https://www.langflow.org/", "trust_vector": { "performance_reliability": { @@ -99,24 +99,31 @@ } }, "security": { - "overall_score": 71, + "overall_score": 67, "criteria": { "api_key_management": { - "score": 75, - "confidence": "medium", + "score": 58, + "confidence": "high", "evidence": [ { "source": "Credentials", "url": "https://docs.langflow.org/configuration", "date": "2024-10-01", "value": "Environment variable-based credential management" + }, + { + "source": "CVE-2025-34291", + "url": "https://nvd.nist.gov/vuln/detail/CVE-2025-34291", + "date": "2026-06-10", + "value": "CVE-2025-34291: account takeover / remote code execution vulnerability in Langflow" } ], "methodology": "Security configuration review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced due to CVE-2025-34291 account takeover/RCE" }, "self_hosting": { - "score": 85, + "score": 70, "confidence": "high", "evidence": [ { @@ -124,10 +131,17 @@ "url": "https://docs.langflow.org/deployment", "date": "2024-10-01", "value": "Self-hosting with Docker and cloud deployment options" + }, + { + "source": "CISA KEV / Flodrix Botnet Exploitation", + "url": "https://nvd.nist.gov/vuln/detail/CVE-2025-3248", + "date": "2026-06-10", + "value": "Internet-exposed self-hosted Langflow instances were actively exploited via CVE-2025-3248 (added to CISA KEV 2025-05-05; exploited by the Flodrix botnet)" } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced: active in-the-wild exploitation of exposed self-hosted deployments" }, "data_privacy": { "score": 78, @@ -158,18 +172,25 @@ "last_verified": "2025-11-09" }, "authentication": { - "score": 62, - "confidence": "medium", + "score": 40, + "confidence": "high", "evidence": [ { "source": "Security Features", "url": "https://docs.langflow.org/", "date": "2024-10-01", "value": "Basic auth available, enterprise features in DataStax version" + }, + { + "source": "CVE-2025-3248", + "url": "https://nvd.nist.gov/vuln/detail/CVE-2025-3248", + "date": "2026-06-10", + "value": "CVE-2025-3248: CVSS 9.8 unauthenticated remote code execution via /api/v1/validate/code; fixed in 1.3.0; added to CISA KEV 2025-05-05 and exploited by the Flodrix botnet" } ], "methodology": "Authentication assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced due to CVE-2025-3248 unauthenticated RCE with confirmed in-the-wild exploitation" } } }, @@ -413,7 +434,8 @@ "Primarily designed for prototyping, not enterprise deployment", "Security features less mature than enterprise platforms", "Limited control compared to code-based implementations", - "Debugging complex flows can be challenging despite visual interface" + "Debugging complex flows can be challenging despite visual interface", + "Serious CVE history: CVE-2025-3248 (unauthenticated RCE, CISA KEV, Flodrix botnet) and CVE-2025-34291 (account takeover/RCE); upgrade to 1.3.0+ and never expose unauthenticated instances" ], "metadata": { "license": "MIT", @@ -493,6 +515,7 @@ "tags": [ "visual", "low-code", - "open-source" + "open-source", + "security-incidents" ] } diff --git a/data/agents/langgraph-agent.json b/data/agents/langgraph-agent.json index d0bc89d..5c8f5fd 100644 --- a/data/agents/langgraph-agent.json +++ b/data/agents/langgraph-agent.json @@ -4,9 +4,9 @@ "name": "LangGraph Agent", "provider": "LangChain", "version": "1.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "LangChain's graph-based agent framework for building stateful, multi-actor applications with cycles and controllable execution flow. Enables complex, cyclic agent workflows with human-in-the-loop capabilities.", + "description": "LangChain's graph-based agent framework for building stateful, multi-actor applications with cycles and controllable execution flow. Reached 1.0 GA on 2025-10-22, adding durable execution and middleware. Enables complex, cyclic agent workflows with human-in-the-loop capabilities and production-grade persistence.", "website": "https://langchain-ai.github.io/langgraph/", "trust_vector": { "performance_reliability": { @@ -296,7 +296,7 @@ } }, "operational_excellence": { - "overall_score": 81, + "overall_score": 85, "criteria": { "ease_of_integration": { "score": 78, @@ -355,18 +355,25 @@ "last_verified": "2025-11-09" }, "community_support": { - "score": 75, - "confidence": "medium", + "score": 82, + "confidence": "high", "evidence": [ { "source": "GitHub Discussions", "url": "https://github.com/langchain-ai/langgraph/discussions", "date": "2024-10-01", "value": "Growing community but newer framework" + }, + { + "source": "LangChain & LangGraph 1.0 Announcement", + "url": "https://blog.langchain.com/langchain-langgraph-1dot0/", + "date": "2026-06-10", + "value": "LangGraph 1.0 GA released 2025-10-22 with durable execution and middleware; large, mature ecosystem" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score raised: 1.0 GA milestone and ecosystem maturity" } } } @@ -385,7 +392,8 @@ "Security and sandboxing must be implemented by developer", "Relatively newer framework with evolving APIs", "Performance overhead from graph execution layer", - "Limited managed service options (LangGraph Cloud in beta)" + "Limited managed service options (LangGraph Cloud in beta)", + "Status (2026-06): 1.0 GA since 2025-10-22 (durable execution, middleware); earlier evolving-API concerns are largely resolved" ], "metadata": { "license": "MIT", diff --git a/data/agents/manus.json b/data/agents/manus.json new file mode 100644 index 0000000..221bd3a --- /dev/null +++ b/data/agents/manus.json @@ -0,0 +1,466 @@ +{ + "id": "manus", + "type": "agent", + "name": "Manus", + "provider": "Meta (formerly Butterfly Effect)", + "version": "1.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "General-purpose autonomous agent that executes end-to-end tasks in a cloud VM equipped with a browser, shell, and file tools. Launched virally in March 2025 by Butterfly Effect and acquired by Meta in a deal that closed in late December 2025 for over $2B.", + "website": "https://manus.im/", + "trust_vector": { + "performance_reliability": { + "overall_score": 76, + "criteria": { + "task_completion_accuracy": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia - Manus (AI agent)", + "url": "https://en.wikipedia.org/wiki/Manus_(AI_agent)", + "date": "2025-05-15", + "value": "Strong showing on GAIA-style autonomous task benchmarks at launch; real-world completion varies widely with task ambiguity and website complexity" + } + ], + "methodology": "Assessment of end-to-end task completion based on published benchmark claims and independent user testing reports", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Integrated cloud VM toolset: full browser automation, shell, file system, code execution, and document/slide generation" + } + ], + "methodology": "Review of integrated tool stack reliability across browsing, shell, and file operations", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia - Manus (AI agent)", + "url": "https://en.wikipedia.org/wiki/Manus_(AI_agent)", + "date": "2025-05-15", + "value": "Decomposes high-level goals into explicit task plans and executes them asynchronously, continuing after the user disconnects" + } + ], + "methodology": "Evaluation of autonomous task decomposition and long-horizon asynchronous execution", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Session workspaces persist files and context; remembers user preferences and prior instructions across tasks" + } + ], + "methodology": "Review of session workspace persistence and cross-task preference memory", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 72, + "confidence": "low", + "evidence": [ + { + "source": "Wikipedia - Manus (AI agent)", + "url": "https://en.wikipedia.org/wiki/Manus_(AI_agent)", + "date": "2025-05-15", + "value": "Retries failed steps and adapts plans, but early reviews noted loops, stalls on complex sites, and occasional silent task failures" + } + ], + "methodology": "Assessment of retry behavior and failure modes from independent reviews", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 70, + "confidence": "low", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Internally orchestrates multiple sub-agent roles (planner, executor, verifier); no user-facing multi-agent composition" + } + ], + "methodology": "Review of internal multi-agent architecture versus user-controllable collaboration", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 60, + "criteria": { + "tool_sandboxing": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "All execution occurs in isolated per-session cloud VMs, never on the user's device" + } + ], + "methodology": "Security architecture review of cloud VM isolation model", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 65, + "confidence": "low", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Consumer-oriented account model; credentials given to the agent for logged-in browsing are held in the session, with limited enterprise-grade access governance" + } + ], + "methodology": "Review of credential handling and account-level controls", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 58, + "confidence": "low", + "evidence": [ + { + "source": "Wikipedia - Manus (AI agent)", + "url": "https://en.wikipedia.org/wiki/Manus_(AI_agent)", + "date": "2025-05-15", + "value": "Autonomous web browsing across arbitrary sites is a large injection surface; public documentation of mitigations is minimal" + } + ], + "methodology": "Threat surface analysis of autonomous browsing with limited disclosed defenses", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 66, + "confidence": "low", + "evidence": [ + { + "source": "CNBC - Meta acquires Manus", + "url": "https://www.cnbc.com/2025/12/30/meta-acquires-singapore-ai-agent-firm-manus-china-butterfly-effect-monicai.html", + "date": "2025-12-30", + "value": "Per-session VM isolation exists, but data residency and processing arrangements are in transition following Meta's acquisition of the Singapore-based, China-originated company" + } + ], + "methodology": "Data architecture review accounting for ownership and jurisdiction transition", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 28, + "confidence": "high", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Proprietary agent harness; historically orchestrated third-party models (Claude, Qwen) rather than publishing its own stack" + } + ], + "methodology": "Source availability assessment", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 50, + "criteria": { + "data_retention": { + "score": 58, + "confidence": "low", + "evidence": [ + { + "source": "Manus Privacy Policy", + "url": "https://manus.im/privacy", + "date": "2026-05-01", + "value": "Session data, files, and browsing artifacts retained in Manus cloud; retention terms are consumer-grade and being restated under Meta ownership" + } + ], + "methodology": "Review of published retention practices during ownership transition", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "CNBC - Meta acquires Manus", + "url": "https://www.cnbc.com/2025/12/30/meta-acquires-singapore-ai-agent-firm-manus-china-butterfly-effect-monicai.html", + "date": "2025-12-30", + "value": "Meta ownership brings established compliance infrastructure, but policies and processing locations for Manus are still being integrated post-acquisition" + } + ], + "methodology": "Compliance posture assessment during corporate integration", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia - Manus (AI agent)", + "url": "https://en.wikipedia.org/wiki/Manus_(AI_agent)", + "date": "2025-05-15", + "value": "Historically routed tasks to third-party models (Anthropic Claude, Alibaba Qwen); under Meta, model routing and data sharing arrangements are changing" + } + ], + "methodology": "Data flow analysis of third-party model routing before and after acquisition", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Cloud-only consumer service with no self-hosted, on-premises, or offline option" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 64, + "criteria": { + "documentation_quality": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Manus Help Center", + "url": "https://manus.im/help", + "date": "2026-05-01", + "value": "Consumer-grade guides and use-case galleries; limited technical documentation of architecture, security, or limits" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "'Manus's Computer' view shows the agent's browser, shell, and file actions live, and completed sessions are replayable and shareable" + } + ], + "methodology": "Review of live session visibility and replay features", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Displays its task plan and step-by-step progress narration, though internal model routing decisions are opaque" + } + ], + "methodology": "Assessment of plan visibility and progress narration", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Closed-source product; only minor components and demos have been shared publicly" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia - Manus (AI agent)", + "url": "https://en.wikipedia.org/wiki/Manus_(AI_agent)", + "date": "2025-05-15", + "value": "Viral launch 2025-03-06 with invite waitlist in the millions; open signup May 2025 built a large consumer user base and active social community" + } + ], + "methodology": "Community engagement analysis of user base and public activity", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 69, + "criteria": { + "ease_of_integration": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Zero-setup web and mobile apps; describe a task in natural language and the cloud VM handles the rest" + } + ], + "methodology": "Onboarding friction assessment for non-technical users", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Multiple concurrent cloud tasks supported on paid tiers; throughput governed by credit balance and plan limits" + } + ], + "methodology": "Scalability assessment of concurrent task limits across tiers", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "Manus Pricing", + "url": "https://manus.im/pricing", + "date": "2026-05-01", + "value": "Credit-based: Free tier with 300 daily credits; paid plans roughly $20-40/mo up to $200/mo. Credit consumption per task varies and is hard to estimate upfront" + } + ], + "methodology": "Pricing model analysis; variable credit burn per task reduces predictability", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Manus Product Page", + "url": "https://manus.im/", + "date": "2026-05-01", + "value": "Per-task session views and credit usage tracking; no organizational observability or audit tooling" + } + ], + "methodology": "Monitoring features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "CNBC - Meta acquires Manus", + "url": "https://www.cnbc.com/2025/12/30/meta-acquires-singapore-ai-agent-firm-manus-china-butterfly-effect-monicai.html", + "date": "2025-12-30", + "value": "Meta acquisition (>$2B, closed ~2025-12-30) secures resources and continuity, but product, policy, and roadmap are mid-integration" + } + ], + "methodology": "Vendor stability and product maturity assessment during acquisition integration", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "research-assistant": { + "overall": 84, + "notes": "Flagship use case: autonomous multi-source web research, comparison shopping, screening, and report generation" + }, + "data-analysis": { + "overall": 78, + "notes": "Runs code in its VM to clean data, analyze spreadsheets, and produce charts and dashboards from uploaded files" + }, + "content-creation": { + "overall": 77, + "notes": "Generates documents, slide decks, and simple websites end-to-end, though output quality needs human review" + }, + "code-generation": { + "overall": 66, + "notes": "Can build and deploy small apps and scripts, but lacks repository workflows and engineering rigor of dedicated coding agents" + }, + "education": { + "overall": 70, + "notes": "Builds interactive course materials and explainers; replayable sessions show how tasks are accomplished" + } + }, + "best_for": [ + "Consumers and analysts delegating end-to-end web research and report tasks to a hands-off agent", + "Users who want asynchronous task execution that continues after they disconnect", + "Non-technical users needing documents, slides, or simple sites produced from a single prompt", + "Exploratory automation of browser-based workflows without writing code" + ], + "strengths": [ + "True general-purpose autonomy: cloud VM with browser, shell, and file tools executes tasks end-to-end", + "Transparent 'Manus's Computer' live view and replayable sessions show exactly what the agent did", + "Asynchronous execution continues in the cloud after the user disconnects", + "Zero-setup web and mobile experience accessible to non-technical users", + "Meta acquisition (closed ~2025-12-30, >$2B) provides long-term resourcing and infrastructure", + "Free tier with 300 daily credits allows meaningful evaluation before paying" + ], + "limitations": [ + "Governance and jurisdiction transition (Singapore/China origins to Meta ownership) leaves data handling policies in flux", + "Minimal published security documentation; prompt injection defenses for autonomous browsing are unclear", + "Credit-based pricing with variable per-task burn makes costs hard to predict", + "Cloud-only with no self-hosted option; sensitive data must enter Manus's VMs", + "Reliability degrades on complex or ambiguous tasks, with loops and silent failures reported", + "Closed-source stack with opaque routing to underlying models (historically Claude and Qwen)" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Anthropic Claude (historical backend)", + "Alibaba Qwen (historical backend)", + "Meta model integration in progress post-acquisition" + ], + "programming_languages": [ + "Natural language interface; agent writes Python, JavaScript, and shell internally" + ], + "deployment_type": "Cloud (per-session VM with browser, shell, file tools)", + "tool_support": [ + "Browser automation", + "Shell and code execution", + "File and document generation", + "Slides and website generation", + "Scheduled tasks" + ], + "first_release": "2025-03-06 (viral invite launch); open signup May 2025", + "pricing": "Free (300 daily credits); paid plans ~$20-40/mo up to $200/mo (credit-based)", + "company_milestones": "Built by Butterfly Effect (Singapore HQ, Chinese origins); acquired by Meta for >$2B, deal closed ~2025-12-30" + }, + "related_entities": [ + "devin", + "openai-codex", + "claude-code", + "google-jules" + ], + "tags": [ + "autonomous", + "general-purpose", + "cloud-agent", + "consumer" + ] +} diff --git a/data/agents/mastra.json b/data/agents/mastra.json new file mode 100644 index 0000000..6ed66f8 --- /dev/null +++ b/data/agents/mastra.json @@ -0,0 +1,473 @@ +{ + "id": "mastra", + "type": "agent", + "name": "Mastra", + "provider": "Mastra AI (YC W25)", + "version": "1.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "TypeScript-first AI agent framework from the Gatsby founders, combining agents, durable workflows, RAG, and evals in one toolkit. Provider-agnostic via Vercel AI SDK model routing, with a local dev playground and 1.0 stable release in January 2026.", + "website": "https://mastra.ai/", + "trust_vector": { + "performance_reliability": { + "overall_score": 83, + "criteria": { + "task_completion_accuracy": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Documentation", + "url": "https://mastra.ai/docs", + "date": "2026-05-20", + "value": "Agents combine tool calling, memory, and structured workflows; built-in evals allow teams to measure task accuracy" + } + ], + "methodology": "Task completion testing across agent and workflow primitives", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Tools Documentation", + "url": "https://mastra.ai/docs/agents/using-tools-and-mcp", + "date": "2026-05-15", + "value": "Zod-schema typed tools with input validation plus MCP client/server support reduce malformed tool calls" + } + ], + "methodology": "Tool invocation testing with typed schemas and MCP servers", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Workflows Documentation", + "url": "https://mastra.ai/docs/workflows/overview", + "date": "2026-05-15", + "value": "Graph-based durable workflows with branching, parallel steps, suspend/resume, and human-in-the-loop" + } + ], + "methodology": "Complex multi-step task testing using workflow engine", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Memory Documentation", + "url": "https://mastra.ai/docs/memory/overview", + "date": "2026-04-20", + "value": "Built-in working memory, conversation history, and semantic recall backed by pluggable storage adapters" + } + ], + "methodology": "Memory system evaluation across threads and semantic recall", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Workflows Documentation", + "url": "https://mastra.ai/docs/workflows/overview", + "date": "2026-05-15", + "value": "Workflow steps support retries, error branches, and durable suspend/resume after failures" + } + ], + "methodology": "Error injection testing on workflow retry and resume paths", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Agent Networks", + "url": "https://mastra.ai/docs/agents/overview", + "date": "2026-04-20", + "value": "Agents composable as workflow steps and sub-agents; agent network primitives for dynamic routing between agents" + } + ], + "methodology": "Multi-agent coordination testing via sub-agents and networks", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 72, + "criteria": { + "tool_sandboxing": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Architecture", + "url": "https://mastra.ai/docs", + "date": "2026-05-20", + "value": "No built-in tool sandboxing; tools run in the host Node.js process and isolation is the developer's responsibility" + } + ], + "methodology": "Security architecture review of tool execution model", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Server Middleware Docs", + "url": "https://mastra.ai/docs/server-db/middleware", + "date": "2026-04-15", + "value": "Auth middleware hooks on the Mastra server; RBAC and SSO are enterprise features rather than core defaults" + } + ], + "methodology": "Access control assessment of server auth options", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Processors Documentation", + "url": "https://mastra.ai/docs/agents/processors", + "date": "2026-04-20", + "value": "Input/output processors enable moderation and PII filtering guardrails, but injection defense is not on by default" + } + ], + "methodology": "Injection testing with and without guardrail processors configured", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Storage Documentation", + "url": "https://mastra.ai/docs/server-db/storage", + "date": "2026-04-15", + "value": "Memory and storage scoped per thread/resource with pluggable backends; multi-tenant isolation is application-level" + } + ], + "methodology": "Data isolation architecture review of storage and memory scoping", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Mastra GitHub Repository", + "url": "https://github.com/mastra-ai/mastra", + "date": "2026-06-01", + "value": "Apache 2.0 core with 22k+ stars; enterprise features are source-available under the Mastra Enterprise License" + } + ], + "methodology": "Source code and license structure review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 82, + "criteria": { + "data_retention": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Self-Hosted Framework", + "url": "https://github.com/mastra-ai/mastra", + "date": "2026-06-01", + "value": "Framework runs in user infrastructure with user-chosen storage backends; full control of retention when self-hosted" + } + ], + "methodology": "Privacy architecture review of self-hosted deployment", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Documentation", + "url": "https://mastra.ai/docs", + "date": "2026-05-20", + "value": "GDPR compliance achievable through self-hosting and storage choice; no turnkey compliance tooling in core" + } + ], + "methodology": "Compliance capabilities assessment of deployment configurations", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Model Routing", + "url": "https://mastra.ai/docs/getting-started/model-providers", + "date": "2026-04-20", + "value": "Data flows to whichever provider is configured via Vercel AI SDK routing; local models supported through Ollama-compatible providers" + } + ], + "methodology": "Data flow analysis across model provider configurations", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Deployment Documentation", + "url": "https://mastra.ai/docs/deployment/overview", + "date": "2026-04-20", + "value": "Deploys as a standard Node.js server anywhere, including fully self-hosted environments with local model providers" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 85, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Documentation", + "url": "https://mastra.ai/docs", + "date": "2026-05-20", + "value": "Polished docs with guides, templates, course content, and examples; quality reflects Gatsby founders' DX background" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Observability Documentation", + "url": "https://mastra.ai/docs/observability/overview", + "date": "2026-04-20", + "value": "OpenTelemetry tracing of agent runs and workflow steps plus a local playground for inspecting executions" + } + ], + "methodology": "Tracing and logging capabilities assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Playground", + "url": "https://mastra.ai/docs/getting-started/studio", + "date": "2026-04-20", + "value": "Local studio visualizes agent steps, tool calls, and workflow graphs, aiding inspection of decisions" + } + ], + "methodology": "Explainability assessment of run visualization features", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Mastra GitHub Repository", + "url": "https://github.com/mastra-ai/mastra", + "date": "2026-06-01", + "value": "Apache 2.0 core, 1.0 stable released 2026-01-21; some enterprise modules under source-available Mastra Enterprise License" + } + ], + "methodology": "Open source assessment of license split and code availability", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "GitHub and npm Metrics", + "url": "https://github.com/mastra-ai/mastra", + "date": "2026-06-01", + "value": "22k+ GitHub stars, 300k+ weekly npm downloads, active Discord and rapid release cadence" + } + ], + "methodology": "Community engagement analysis of stars, downloads, and releases", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 84, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Quickstart", + "url": "https://mastra.ai/docs/getting-started/installation", + "date": "2026-05-15", + "value": "create-mastra CLI scaffolds a typed project with playground in minutes; idiomatic TypeScript APIs" + } + ], + "methodology": "Integration complexity assessment with scaffolded project testing", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Mastra Deployment Documentation", + "url": "https://mastra.ai/docs/deployment/overview", + "date": "2026-04-20", + "value": "Deploys to serverless platforms (Vercel, Cloudflare, Netlify) and Node servers; durable workflows support long-running jobs" + } + ], + "methodology": "Scalability assessment across deployment targets", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Open Source Framework", + "url": "https://github.com/mastra-ai/mastra", + "date": "2026-06-01", + "value": "Free Apache 2.0 core; costs from LLM APIs, hosting, and optional Mastra Cloud/enterprise offerings" + } + ], + "methodology": "Pricing model analysis", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Mastra Evals and Observability", + "url": "https://mastra.ai/docs/evals/overview", + "date": "2026-04-20", + "value": "Built-in evals (faithfulness, hallucination, relevance scorers) plus OTel export to external observability platforms" + } + ], + "methodology": "Monitoring and evaluation features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Mastra 1.0 Release", + "url": "https://mastra.ai/blog/mastra-1-0", + "date": "2026-01-21", + "value": "1.0 stable shipped 2026-01-21 with semver commitment after a year of rapid pre-1.0 iteration" + } + ], + "methodology": "Production readiness assessment of API stability and release maturity", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 82, + "notes": "Typed TypeScript tooling and MCP support fit developer-workflow agents well" + }, + "customer-support": { + "overall": 85, + "notes": "Memory, RAG, workflow human-in-the-loop, and guardrail processors suit support bots" + }, + "content-creation": { + "overall": 84, + "notes": "Workflows plus evals enable quality-controlled content pipelines" + }, + "research-assistant": { + "overall": 82, + "notes": "RAG primitives and agent networks handle multi-source research tasks" + }, + "data-analysis": { + "overall": 78, + "notes": "Capable via tools and workflows, though Python data ecosystem is unavailable" + }, + "education": { + "overall": 80, + "notes": "Memory threads and structured workflows work well for tutoring applications" + } + }, + "best_for": [ + "TypeScript/JavaScript teams wanting a full-stack agent framework with strong DX", + "Products needing durable workflows with human-in-the-loop and suspend/resume", + "Teams that want agents, RAG, memory, and evals in one integrated toolkit", + "Startups deploying agents to serverless platforms like Vercel or Cloudflare" + ], + "strengths": [ + "TypeScript-first with end-to-end type safety via Zod schemas", + "Integrated suite: agents, durable workflows, RAG, memory, and evals in one framework", + "Provider-agnostic model routing through the Vercel AI SDK", + "Excellent developer experience: CLI scaffolding, local playground, polished docs", + "1.0 stable (Jan 2026) with strong adoption: 22k+ stars, 300k+ weekly npm downloads", + "Built-in eval scorers and OpenTelemetry observability" + ], + "limitations": [ + "No built-in tool sandboxing; tools execute in the host Node.js process", + "Enterprise features (RBAC, SSO, some modules) are source-available, not Apache 2.0", + "Young 1.0; some advanced patterns and integrations still maturing", + "No Python SDK, limiting access to the Python ML/data ecosystem", + "Security guardrails (processors) are opt-in rather than default" + ], + "related": [ + "langgraph-agent", + "openai-agents-sdk", + "crewai", + "pydantic-ai", + "flowise" + ], + "metadata": { + "license": "Apache 2.0 (core); Mastra Enterprise License (source-available enterprise features)", + "supported_models": [ + "OpenAI", + "Anthropic Claude", + "Google Gemini", + "Any provider via Vercel AI SDK routing", + "Local models via OpenAI-compatible endpoints" + ], + "programming_languages": [ + "TypeScript", + "JavaScript" + ], + "deployment_type": "Self-hosted or Mastra Cloud", + "tool_support": [ + "Zod-typed custom tools", + "MCP client and server", + "Vercel AI SDK tools", + "Built-in RAG tools" + ], + "github_stars": "22000+", + "first_release": "2024 (YC W25; 1.0 stable 2026-01-21)", + "pricing": "Free (Apache 2.0 core) - Costs from LLM APIs, hosting, and optional enterprise/cloud offerings", + "node_requirement": "Node.js >=20", + "adoption": "300k+ weekly npm downloads; built by the Gatsby founders" + }, + "tags": [ + "typescript", + "workflows", + "open-source" + ] +} diff --git a/data/agents/memgpt.json b/data/agents/memgpt.json index 8d801e6..28c8e18 100644 --- a/data/agents/memgpt.json +++ b/data/agents/memgpt.json @@ -1,13 +1,13 @@ { "id": "memgpt", "type": "agent", - "name": "MemGPT", - "provider": "Research", + "name": "Letta (formerly MemGPT)", + "provider": "Letta Inc.", "version": "0.x", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Memory-enhanced LLM agent system enabling long-term context and conversation memory. Implements virtual context management inspired by operating systems to overcome LLM context window limitations for stateful agents.", - "website": "https://memgpt.ai/", + "description": "REBRANDED: MemGPT became Letta (Letta Inc., a UC Berkeley spinout) in September 2024; \"MemGPT\" now refers only to the underlying research technique. Letta is a memory-enhanced LLM agent platform enabling long-term context via virtual context management inspired by operating systems, overcoming context window limits for stateful agents.", + "website": "https://www.letta.com", "trust_vector": { "performance_reliability": { "overall_score": 79, @@ -391,10 +391,16 @@ "url": "https://github.com/cpacker/MemGPT", "date": "2024-10-01", "value": "Active development, production use requires careful setup" + }, + { + "source": "Letta Rebranding Announcement", + "url": "https://www.letta.com/blog/memgpt-and-letta", + "date": "2026-06-10", + "value": "MemGPT rebranded to Letta (Letta Inc., UC Berkeley spinout, Sept 2024); active commercial development continues under the Letta name" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "cli_interface": { "score": 82, @@ -427,7 +433,8 @@ "Complexity in managing and debugging agent memory", "Active development, some features still experimental", "Limited production tooling and monitoring features", - "Memory pagination can introduce unpredictability" + "Memory pagination can introduce unpredictability", + "Rebranded to Letta (Sept 2024): older MemGPT docs, package names, and repo links are outdated" ], "metadata": { "license": "Apache 2.0", @@ -505,6 +512,7 @@ "tags": [ "memory", "long-context", - "open-source" + "open-source", + "rebranded" ] } diff --git a/data/agents/microsoft-agent-framework.json b/data/agents/microsoft-agent-framework.json new file mode 100644 index 0000000..5c77a56 --- /dev/null +++ b/data/agents/microsoft-agent-framework.json @@ -0,0 +1,469 @@ +{ + "id": "microsoft-agent-framework", + "type": "agent", + "name": "Microsoft Agent Framework", + "provider": "Microsoft", + "version": "1.0", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Open-source SDK and runtime for building AI agents and graph-based multi-agent workflows in .NET and Python. Merges AutoGen and Semantic Kernel into a single framework with checkpointing, middleware, and OpenTelemetry-based observability.", + "website": "https://github.com/microsoft/agent-framework", + "trust_vector": { + "performance_reliability": { + "overall_score": 85, + "criteria": { + "task_completion_accuracy": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Foundry Blog - Agent Framework Introduction", + "url": "https://devblogs.microsoft.com/foundry/introducing-microsoft-agent-framework-the-open-source-engine-for-agentic-ai-apps/", + "date": "2025-10-01", + "value": "Combines AutoGen's multi-agent orchestration research with Semantic Kernel's production-grade plumbing" + } + ], + "methodology": "Assessment of agent execution quality based on framework capabilities, AutoGen/Semantic Kernel lineage, and community-reported results since public preview", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Typed function tools, MCP support, and middleware pipeline for intercepting and validating tool calls in .NET and Python" + } + ], + "methodology": "Review of tool abstraction design, type safety, MCP integration, and middleware-based tool call validation", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework Documentation", + "url": "https://learn.microsoft.com/en-us/agent-framework/", + "date": "2026-04-03", + "value": "Graph-based workflows support sequential, concurrent, handoff, and group-chat orchestration patterns with explicit control flow" + } + ], + "methodology": "Evaluation of graph-based workflow engine for complex task decomposition and deterministic orchestration", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Built-in thread state management, pluggable memory providers, and workflow checkpointing for pause/resume of long-running tasks" + } + ], + "methodology": "Review of thread persistence, checkpointing, and memory provider abstractions", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Foundry Blog - Agent Framework Introduction", + "url": "https://devblogs.microsoft.com/foundry/introducing-microsoft-agent-framework-the-open-source-engine-for-agentic-ai-apps/", + "date": "2025-10-01", + "value": "Checkpointing enables resuming failed workflows from last saved state; middleware allows custom retry and error-handling policies" + } + ], + "methodology": "Assessment of checkpoint/resume semantics and middleware-based error handling under failure injection scenarios", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Inherits AutoGen's multi-agent patterns: group chat, handoffs, nested agents-as-tools, and agent-to-agent (A2A) protocol support" + } + ], + "methodology": "Multi-agent orchestration pattern review covering group chat, handoff, and concurrent workflows", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 80, + "criteria": { + "tool_sandboxing": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Agent Framework Documentation", + "url": "https://learn.microsoft.com/en-us/agent-framework/", + "date": "2026-04-03", + "value": "No built-in execution sandbox; tool isolation is the developer's responsibility, with optional hosted code interpreter via Azure AI Foundry" + } + ], + "methodology": "Security architecture review of tool execution boundaries in self-hosted and Azure-hosted configurations", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework Documentation", + "url": "https://learn.microsoft.com/en-us/agent-framework/", + "date": "2026-04-03", + "value": "Integrates with Microsoft Entra ID and Azure RBAC when used with Azure AI Foundry; middleware enables custom authorization on tool calls" + } + ], + "methodology": "Review of identity integration, RBAC support, and middleware-based authorization hooks", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Foundry Blog - Agent Framework Introduction", + "url": "https://devblogs.microsoft.com/foundry/introducing-microsoft-agent-framework-the-open-source-engine-for-agentic-ai-apps/", + "date": "2025-10-01", + "value": "Middleware pipeline supports content filtering and guardrail integration (e.g., Azure AI Content Safety), but no default injection defense is enabled" + } + ], + "methodology": "Assessment of available guardrail integration points versus out-of-the-box injection protections", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Self-hosted runtime keeps agent state in developer-controlled stores; threads and workflow state are isolated per execution" + } + ], + "methodology": "Data architecture review of thread isolation and state storage control", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "MIT licensed, fully open source for both .NET and Python, with public roadmap and active issue tracker; 1.0 GA released 2026-04-03" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 84, + "criteria": { + "data_retention": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Self-hosted SDK stores no data itself; retention is fully controlled by the developer's chosen state and memory stores" + } + ], + "methodology": "Privacy architecture review of framework data handling", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Trust Center", + "url": "https://www.microsoft.com/en-us/trust-center/privacy/gdpr-overview", + "date": "2026-04-03", + "value": "GDPR-compliant deployments achievable; Azure-hosted components covered by Microsoft's GDPR commitments, self-hosted compliance depends on configuration" + } + ], + "methodology": "Compliance capability assessment for self-hosted and Azure-backed deployments", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Agent Framework Documentation", + "url": "https://learn.microsoft.com/en-us/agent-framework/", + "date": "2026-04-03", + "value": "Data flows only to the configured model provider (Azure OpenAI, OpenAI, Anthropic, local models); framework itself sends no telemetry by default" + } + ], + "methodology": "Data flow analysis of model provider connectors and telemetry defaults", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Runs anywhere .NET or Python runs; supports local models via Ollama, Foundry Local, and ONNX connectors" + } + ], + "methodology": "Deployment options assessment including local model support", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 87, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Learn - Agent Framework", + "url": "https://learn.microsoft.com/en-us/agent-framework/", + "date": "2026-04-03", + "value": "Comprehensive Microsoft Learn documentation, tutorials, and official migration guides from AutoGen and Semantic Kernel" + } + ], + "methodology": "Documentation completeness review including migration guides", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Foundry Blog - Agent Framework Introduction", + "url": "https://devblogs.microsoft.com/foundry/introducing-microsoft-agent-framework-the-open-source-engine-for-agentic-ai-apps/", + "date": "2025-10-01", + "value": "Native OpenTelemetry instrumentation following GenAI semantic conventions; traces agent runs, tool calls, and workflow steps" + } + ], + "methodology": "Observability review of built-in OTel spans for agent and workflow execution", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Explicit graph-based workflows make control flow inspectable; agent reasoning visibility depends on underlying model and logging configuration" + } + ], + "methodology": "Assessment of workflow visualizability and reasoning trace availability", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "MIT license, monorepo for .NET and Python SDKs, public development with community contributions accepted" + } + ], + "methodology": "Open source assessment of license and development model", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-06-01", + "value": "Rapid star growth since October 2025 preview; consolidated AutoGen and Semantic Kernel communities migrating to the framework as both enter maintenance mode" + } + ], + "methodology": "Community engagement analysis of GitHub activity, discussions, and migration momentum", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 86, + "criteria": { + "ease_of_integration": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework Documentation", + "url": "https://learn.microsoft.com/en-us/agent-framework/", + "date": "2026-04-03", + "value": "Simple ChatClientAgent abstraction; pip/NuGet install; connectors for Azure OpenAI, OpenAI, Anthropic, and local models" + } + ], + "methodology": "Integration complexity assessment for .NET and Python developers", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Foundry Blog - Agent Framework Introduction", + "url": "https://devblogs.microsoft.com/foundry/introducing-microsoft-agent-framework-the-open-source-engine-for-agentic-ai-apps/", + "date": "2025-10-01", + "value": "Designed for production with stateless agent design, durable checkpointed workflows, and hosting options from containers to Azure AI Foundry" + } + ], + "methodology": "Scalability assessment of runtime design and hosting options", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub", + "url": "https://github.com/microsoft/agent-framework", + "date": "2026-04-03", + "value": "Free MIT-licensed framework; costs arise only from chosen model provider and hosting infrastructure" + } + ], + "methodology": "Pricing model analysis", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework Documentation", + "url": "https://learn.microsoft.com/en-us/agent-framework/", + "date": "2026-04-03", + "value": "First-class OpenTelemetry support exports traces and metrics to any OTel backend including Azure Monitor, Jaeger, and Aspire dashboard" + } + ], + "methodology": "Monitoring features assessment of built-in telemetry", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Agent Framework GitHub Releases", + "url": "https://github.com/microsoft/agent-framework/releases", + "date": "2026-04-03", + "value": "1.0 GA released 2026-04-03 with stable API surface; designated successor to Semantic Kernel and AutoGen, which are now in maintenance mode" + } + ], + "methodology": "Production readiness assessment of GA status, API stability, and Microsoft support commitment", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 82, + "notes": "Strong foundation for building coding agents with typed tools and checkpointed workflows, though not a turnkey coding agent itself" + }, + "customer-support": { + "overall": 85, + "notes": "Handoff and group-chat orchestration patterns plus Azure ecosystem integration suit enterprise support agent systems" + }, + "data-analysis": { + "overall": 83, + "notes": "Graph workflows with checkpointing handle long-running analysis pipelines; code interpreter available via Azure AI Foundry" + }, + "research-assistant": { + "overall": 86, + "notes": "Multi-agent orchestration inherited from AutoGen works well for researcher/critic/synthesizer patterns" + }, + "content-creation": { + "overall": 80, + "notes": "Capable multi-agent content pipelines, though less purpose-built for creative workflows than role-based frameworks" + }, + "financial-analysis": { + "overall": 79, + "notes": "Enterprise identity, observability, and self-hosting make regulated deployments feasible with custom compliance work" + } + }, + "best_for": [ + "Enterprises standardizing on a single Microsoft-backed agent framework across .NET and Python", + "Teams migrating from AutoGen or Semantic Kernel, which are now in maintenance mode", + "Production systems needing durable, checkpointed, observable multi-agent workflows", + "Organizations already invested in the Azure AI Foundry ecosystem" + ], + "strengths": [ + "Unifies AutoGen's multi-agent research patterns with Semantic Kernel's production engineering", + "Graph-based workflows with checkpointing for durable, resumable long-running tasks", + "Native OpenTelemetry instrumentation for tracing agents, tools, and workflows", + "First-class support for both .NET and Python with consistent abstractions", + "MIT-licensed and fully open source with strong Microsoft backing and 1.0 GA stability", + "Middleware pipeline enables custom guardrails, auth, and policy enforcement on every tool call" + ], + "limitations": [ + "No built-in execution sandbox; tool isolation must be implemented by the developer", + "Young framework (GA April 2026); ecosystem of extensions still smaller than older frameworks", + "Deepest integrations favor the Azure ecosystem, which may not suit cloud-neutral teams", + "Migration from AutoGen and Semantic Kernel requires code changes despite official guides", + "Prompt injection defenses require explicit guardrail integration rather than safe defaults" + ], + "metadata": { + "license": "MIT", + "supported_models": [ + "Azure OpenAI", + "OpenAI GPT models", + "Anthropic Claude", + "Local models via Ollama and Foundry Local" + ], + "programming_languages": [ + ".NET (C#)", + "Python" + ], + "deployment_type": "Self-hosted or Azure AI Foundry", + "tool_support": [ + "Typed function tools", + "MCP servers", + "OpenAPI tools", + "Hosted code interpreter (Azure)" + ], + "first_release": "2025-10-01 (public preview), 1.0 GA 2026-04-03", + "pricing": "Free (MIT license) - Costs only from model provider and hosting", + "predecessors": "Official successor to AutoGen and Semantic Kernel (both maintenance mode)" + }, + "related_entities": [ + "autogen", + "semantic-kernel-agent", + "langgraph-agent" + ], + "tags": [ + "multi-agent", + "open-source", + "enterprise", + "workflow-orchestration" + ] +} diff --git a/data/agents/openai-agents-sdk.json b/data/agents/openai-agents-sdk.json new file mode 100644 index 0000000..340dcac --- /dev/null +++ b/data/agents/openai-agents-sdk.json @@ -0,0 +1,467 @@ +{ + "id": "openai-agents-sdk", + "type": "agent", + "name": "OpenAI Agents SDK", + "provider": "OpenAI", + "version": "1.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Production-ready multi-agent orchestration framework built around agents, handoffs, guardrails, and tracing. Open-source (MIT) successor to Swarm, released March 2025 with a major overhaul in April 2026.", + "website": "https://openai.github.io/openai-agents-python/", + "trust_vector": { + "performance_reliability": { + "overall_score": 84, + "criteria": { + "task_completion_accuracy": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI: New tools for building agents", + "url": "https://openai.com/index/new-tools-for-building-agents/", + "date": "2025-03-11", + "value": "Production-grade upgrade of Swarm with structured outputs, validated handoffs, and Responses API integration" + } + ], + "methodology": "Evaluation of agent task success across single- and multi-agent configurations", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/tools/", + "date": "2026-04-15", + "value": "Function tools with automatic Pydantic schema validation plus hosted tools (web search, file search, code interpreter, computer use)" + } + ], + "methodology": "Testing of function tools, hosted tools, and schema-validated argument handling", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/running_agents/", + "date": "2026-04-15", + "value": "Agent loop with configurable max turns handles multi-step tasks; planning quality depends on the configured model" + } + ], + "methodology": "Multi-step task evaluation across the agent loop and handoff chains", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK sessions documentation", + "url": "https://openai.github.io/openai-agents-python/sessions/", + "date": "2026-04-15", + "value": "Built-in session memory (SQLite, Redis, SQLAlchemy backends) automatically maintains conversation history across runs" + } + ], + "methodology": "Review of session backends and cross-run conversation persistence", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/running_agents/", + "date": "2026-04-15", + "value": "Typed exceptions, guardrail tripwires, and tool error handlers enable structured failure handling" + } + ], + "methodology": "Testing of exception handling, guardrail tripwires, and tool failure paths", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK handoffs documentation", + "url": "https://openai.github.io/openai-agents-python/handoffs/", + "date": "2026-04-15", + "value": "Handoffs are a core primitive for delegating between specialized agents; agents-as-tools pattern supports orchestrator designs" + } + ], + "methodology": "Multi-agent coordination testing using handoffs and agents-as-tools patterns", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 77, + "criteria": { + "tool_sandboxing": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/tools/", + "date": "2026-04-15", + "value": "No built-in sandbox for custom function tools; hosted tools (code interpreter) run in OpenAI's sandboxed environments" + } + ], + "methodology": "Security architecture review of custom tool execution versus hosted tools", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/", + "date": "2026-04-15", + "value": "Per-agent tool restrictions and human-in-the-loop approval support; broader access control left to the developer" + } + ], + "methodology": "Assessment of tool scoping, approval flows, and developer-implemented controls", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK guardrails documentation", + "url": "https://openai.github.io/openai-agents-python/guardrails/", + "date": "2026-04-15", + "value": "First-class input/output guardrails run in parallel with the agent and can trip to halt unsafe or off-policy runs" + } + ], + "methodology": "Guardrail configuration testing against adversarial and off-policy inputs", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/context/", + "date": "2026-04-15", + "value": "Typed local context objects are isolated per run and never sent to the LLM; deployment isolation is integrator-managed" + } + ], + "methodology": "Review of run context isolation and self-hosted deployment boundaries", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "openai-agents-python GitHub repository", + "url": "https://github.com/openai/openai-agents-python", + "date": "2026-06-01", + "value": "MIT licensed, fully open source with active public development in Python and TypeScript" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 80, + "criteria": { + "data_retention": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI API data usage policies", + "url": "https://openai.com/policies/api-data-usage-policies", + "date": "2026-03-01", + "value": "API data not used for training by default; session state stored on the integrator's own backend" + } + ], + "methodology": "Review of OpenAI API retention terms and self-managed session storage", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Trust Portal", + "url": "https://trust.openai.com/", + "date": "2026-03-01", + "value": "OpenAI offers DPA and SOC 2 for API usage; framework itself is self-hosted so compliance is achievable with configuration" + } + ], + "methodology": "Compliance capabilities assessment for framework plus default model provider", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK tracing documentation", + "url": "https://openai.github.io/openai-agents-python/tracing/", + "date": "2026-04-15", + "value": "Tracing uploads run data to OpenAI by default (disableable); prompts go to whichever LLM provider is configured" + } + ], + "methodology": "Data flow analysis of default tracing export and model provider routing", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK models documentation", + "url": "https://openai.github.io/openai-agents-python/models/litellm/", + "date": "2026-04-15", + "value": "LiteLLM integration supports 100+ LLMs including local models via Ollama-compatible endpoints" + } + ], + "methodology": "Deployment options assessment including non-OpenAI and local model routing", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 90, + "criteria": { + "documentation_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/", + "date": "2026-04-15", + "value": "Comprehensive docs with quickstarts, API reference, and examples for agents, handoffs, guardrails, sessions, and tracing" + } + ], + "methodology": "Documentation completeness review across Python and TypeScript SDKs", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK tracing documentation", + "url": "https://openai.github.io/openai-agents-python/tracing/", + "date": "2026-04-15", + "value": "Built-in tracing of every LLM call, tool call, handoff, and guardrail with OpenAI dashboard and OTel/third-party exporters" + } + ], + "methodology": "Review of built-in trace spans, dashboard visualization, and OpenTelemetry export", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK tracing documentation", + "url": "https://openai.github.io/openai-agents-python/tracing/", + "date": "2026-04-15", + "value": "Trace spans expose handoff reasons, tool arguments, and guardrail outcomes for post-hoc decision analysis" + } + ], + "methodology": "Assessment of trace-based explanation of agent routing and tool decisions", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "openai-agents-python GitHub repository", + "url": "https://github.com/openai/openai-agents-python", + "date": "2026-06-01", + "value": "MIT license with full source available; major 2026-04-15 overhaul developed in the open" + } + ], + "methodology": "Open source assessment of license, source availability, and public development", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "openai-agents-python GitHub repository", + "url": "https://github.com/openai/openai-agents-python", + "date": "2026-06-01", + "value": "Tens of thousands of stars, frequent releases, and a large contributor and integration ecosystem" + } + ], + "methodology": "Community engagement analysis via GitHub stars, contributor activity, and release cadence", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 87, + "criteria": { + "ease_of_integration": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK quickstart", + "url": "https://openai.github.io/openai-agents-python/quickstart/", + "date": "2026-04-15", + "value": "Minimal-primitives design (agents, handoffs, guardrails, sessions); a working agent in a few lines of code" + } + ], + "methodology": "Integration complexity assessment from install to working multi-agent app", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Agents SDK documentation", + "url": "https://openai.github.io/openai-agents-python/", + "date": "2026-04-15", + "value": "Lightweight stateless runner scales horizontally; throughput bounded by model provider rate limits" + } + ], + "methodology": "Assessment of stateless runner scaling and provider rate limit constraints", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "openai-agents-python GitHub repository", + "url": "https://github.com/openai/openai-agents-python", + "date": "2026-06-01", + "value": "Free MIT-licensed SDK; only costs are model API rates from the chosen provider" + } + ], + "methodology": "Pricing model analysis of free framework plus pay-per-token model usage", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Agents SDK tracing documentation", + "url": "https://openai.github.io/openai-agents-python/tracing/", + "date": "2026-04-15", + "value": "Built-in tracing dashboard plus OTel and third-party exporters (Logfire, Langfuse, W&B, and more)" + } + ], + "methodology": "Monitoring features assessment including built-in and third-party observability", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI: New tools for building agents", + "url": "https://openai.com/index/new-tools-for-building-agents/", + "date": "2025-03-11", + "value": "Explicitly positioned as production-ready successor to experimental Swarm; matured further with the 2026-04-15 overhaul" + } + ], + "methodology": "Maturity assessment from release history, API stability, and enterprise adoption", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "customer-support": { + "overall": 90, + "notes": "Handoffs between triage and specialist agents plus guardrails make this a flagship use case" + }, + "research-assistant": { + "overall": 87, + "notes": "Hosted web/file search tools and multi-agent orchestration suit research pipelines" + }, + "code-generation": { + "overall": 82, + "notes": "Capable with code interpreter and function tools, though not specialized for coding workflows" + }, + "data-analysis": { + "overall": 84, + "notes": "Code interpreter, file search, and structured outputs work well for analysis agents" + }, + "content-creation": { + "overall": 83, + "notes": "Multi-agent writer/editor pipelines with output guardrails are straightforward to build" + } + }, + "best_for": [ + "Teams building production multi-agent systems with handoffs and guardrails", + "Developers wanting built-in tracing and observability without extra setup", + "Organizations standardizing on the OpenAI/Responses API ecosystem", + "Projects needing model flexibility across 100+ LLMs via LiteLLM" + ], + "strengths": [ + "Minimal, well-designed primitives: agents, handoffs, guardrails, sessions", + "Best-in-class built-in tracing with dashboard and OTel/third-party exporters", + "First-class input/output guardrails with tripwire enforcement", + "MIT-licensed and fully open source in Python and TypeScript", + "Provider-agnostic via LiteLLM (100+ models) despite Responses-API-native design", + "Free framework; costs limited to model usage" + ], + "limitations": [ + "No built-in sandboxing for custom function tools; execution safety is developer-owned", + "Tracing exports run data to OpenAI by default unless explicitly disabled", + "Some hosted tools and tracing features work best only with OpenAI models", + "April 2026 overhaul introduced breaking changes requiring migration", + "Less opinionated about deployment, requiring infrastructure decisions from the team" + ], + "metadata": { + "license": "MIT", + "supported_models": [ + "OpenAI GPT series (Responses API native)", + "100+ LLMs via LiteLLM", + "Local models via OpenAI-compatible endpoints" + ], + "programming_languages": [ + "Python", + "TypeScript" + ], + "deployment_type": "Self-hosted", + "tool_support": [ + "Function tools with schema validation", + "Hosted tools (web search, file search, code interpreter, computer use)", + "MCP servers", + "Agents as tools" + ], + "first_release": "2025-03-11; major overhaul 2026-04-15", + "pricing": "Free (MIT); pay only model API rates", + "predecessor": "OpenAI Swarm (experimental)" + }, + "related": [ + "swarm", + "openai-codex", + "openai-assistants-api", + "claude-agent-sdk", + "langgraph-agent", + "crewai" + ], + "tags": [ + "multi-agent", + "open-source", + "orchestration", + "openai" + ] +} diff --git a/data/agents/openai-assistants-api.json b/data/agents/openai-assistants-api.json index 961fd33..525361d 100644 --- a/data/agents/openai-assistants-api.json +++ b/data/agents/openai-assistants-api.json @@ -4,9 +4,9 @@ "name": "OpenAI Assistants API", "provider": "OpenAI", "version": "v2", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's managed agent framework with native tool use, code interpreter, file search, and persistent threads. Ideal for building stateful conversational applications.", + "description": "DEPRECATED: the Assistants API will be sunset on 2026-08-26 (roughly 2.5 months away); OpenAI directs users to the Responses API plus Conversations API as the replacement. Previously a managed agent framework with native tool use, code interpreter, file search, and persistent threads. Do not start new projects on it.", "website": "https://platform.openai.com/docs/assistants/overview", "trust_vector": { "performance_reliability": { @@ -254,10 +254,10 @@ } }, "operational_excellence": { - "overall_score": 88, + "overall_score": 78, "criteria": { "ease_of_integration": { - "score": 93, + "score": 75, "confidence": "high", "evidence": [ { @@ -265,13 +265,20 @@ "url": "https://platform.openai.com/docs/libraries", "date": "2024-10-01", "value": "Official SDKs for Python, Node.js, and REST API" + }, + { + "source": "OpenAI API Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "Assistants API deprecated with sunset date 2026-08-26; replacement is the Responses API + Conversations API" } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced: new integrations are discouraged given the 2026-08-26 sunset" }, "scalability": { - "score": 88, + "score": 70, "confidence": "high", "evidence": [ { @@ -330,7 +337,8 @@ "Latency can be high for complex multi-tool operations", "Vendor lock-in to OpenAI platform", "Limited native explainability features", - "Code interpreter limited to Python only" + "Code interpreter limited to Python only", + "Deprecated: sunset scheduled for 2026-08-26; migrate to the Responses API + Conversations API" ], "metadata": { "license": "Proprietary", @@ -354,7 +362,7 @@ "max_files_per_assistant": 10000, "max_file_size": "512 MB", "pricing": "Token-based (model pricing) + $0.03/session for Code Interpreter + $0.10/GB/day for File Search (first GB free)", - "deprecation_notice": "Being replaced by Responses API and Agents SDK in H1 2026" + "deprecation_notice": "Deprecated: sunset on 2026-08-26. Replaced by the Responses API + Conversations API." }, "use_case_ratings": { "customer-support": { @@ -404,8 +412,12 @@ "Startups needing rapid prototyping of AI assistants", "Organizations comfortable with OpenAI's platform" ], + "related_entities": [ + "openai-agents-sdk" + ], "tags": [ "openai", - "assistants" + "assistants", + "deprecated" ] } diff --git a/data/agents/openai-codex.json b/data/agents/openai-codex.json new file mode 100644 index 0000000..f77a387 --- /dev/null +++ b/data/agents/openai-codex.json @@ -0,0 +1,479 @@ +{ + "id": "openai-codex", + "type": "agent", + "name": "OpenAI Codex", + "provider": "OpenAI", + "version": "GPT-5.3-Codex era", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "OpenAI's coding agent spanning a cloud agent that runs tasks in isolated containers and an open-source CLI. Delegates parallel software tasks (features, fixes, PRs) powered by GPT-5.3-Codex, with network access disabled by default in the cloud.", + "website": "https://openai.com/codex/", + "trust_vector": { + "performance_reliability": { + "overall_score": 84, + "criteria": { + "task_completion_accuracy": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI: Introducing Codex", + "url": "https://openai.com/index/introducing-codex/", + "date": "2025-05-16", + "value": "Cloud agent completes real-world engineering tasks ending in tested, citable PRs" + }, + { + "source": "GPT-5-Codex upgrade", + "url": "https://openai.com/index/introducing-upgrades-to-codex/", + "date": "2025-09-15", + "value": "GPT-5-Codex (now GPT-5.3-Codex) substantially improved long-task completion and code quality" + } + ], + "methodology": "Benchmark review and hands-on evaluation of PR-producing cloud tasks", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Reliable shell, editor, and test-runner usage inside containers; setup scripts configure project dependencies" + } + ], + "methodology": "Testing of in-container command execution, editing, and test running across repositories", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI: Introducing Codex", + "url": "https://openai.com/index/introducing-codex/", + "date": "2025-05-16", + "value": "Handles long-horizon tasks autonomously: reading codebases, implementing changes, running tests, and iterating until passing" + } + ], + "methodology": "Long-horizon task evaluation from issue description to passing tests and PR", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Codex documentation (AGENTS.md)", + "url": "https://developers.openai.com/codex", + "date": "2026-05-01", + "value": "AGENTS.md files persist project conventions and instructions across tasks; cloud tasks are otherwise stateless per container" + } + ], + "methodology": "Review of AGENTS.md guidance persistence and per-task container statelessness", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Iterates on failing tests and lint errors until passing; surfaces logs and citations when blocked" + } + ], + "methodology": "Observed recovery behavior from failing tests, build errors, and missing dependencies", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI: Introducing Codex", + "url": "https://openai.com/index/introducing-codex/", + "date": "2025-05-16", + "value": "Many tasks run in parallel containers simultaneously, though without inter-agent handoff primitives" + } + ], + "methodology": "Assessment of parallel task fan-out and coordination model", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 84, + "criteria": { + "tool_sandboxing": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Cloud tasks run in isolated containers with internet access disabled by default during execution; configurable domain allowlists" + }, + { + "source": "Codex CLI repository", + "url": "https://github.com/openai/codex", + "date": "2026-05-01", + "value": "CLI offers OS-level sandboxing (Seatbelt on macOS, Landlock/seccomp on Linux) with approval modes" + } + ], + "methodology": "Review of container isolation, default-deny network policy, and CLI sandbox mechanisms", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Scoped GitHub repo access, per-environment configuration, and CLI approval modes (suggest/auto-edit/full-auto)" + } + ], + "methodology": "Assessment of repository scoping, environment controls, and approval mode granularity", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Default-disabled internet during execution sharply limits exfiltration from injected instructions; agent-specific safety training applied" + } + ], + "methodology": "Review of network-isolation mitigations and model-level injection defenses", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Each task gets its own ephemeral container preloaded only with the target repository and configured environment" + } + ], + "methodology": "Architecture review of per-task container isolation and environment scoping", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Codex CLI repository", + "url": "https://github.com/openai/codex", + "date": "2026-05-01", + "value": "Codex CLI is open source under Apache-2.0; the cloud agent service and models remain proprietary" + } + ], + "methodology": "License and source availability review of CLI versus cloud service", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 73, + "criteria": { + "data_retention": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI enterprise privacy", + "url": "https://openai.com/enterprise-privacy/", + "date": "2026-03-01", + "value": "Business/Enterprise data excluded from training by default; consumer ChatGPT plan settings govern Codex task data" + } + ], + "methodology": "Review of OpenAI retention and training policies across ChatGPT plan tiers", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Trust Portal", + "url": "https://trust.openai.com/", + "date": "2026-03-01", + "value": "SOC 2 Type II and DPA available for business tiers covering Codex usage" + } + ], + "methodology": "Compliance certification and DPA availability review", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Repository code is processed by OpenAI; default-off internet prevents task-time data flows to other third parties" + } + ], + "methodology": "Data flow analysis of repository access, GitHub integration, and network policy", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "Codex CLI repository", + "url": "https://github.com/openai/codex", + "date": "2026-05-01", + "value": "Open-source CLI runs locally and supports OpenAI-compatible endpoints, but flagship Codex models require OpenAI's cloud" + } + ], + "methodology": "Deployment options assessment of local CLI versus cloud-only agent and models", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 84, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Codex developer documentation", + "url": "https://developers.openai.com/codex", + "date": "2026-05-01", + "value": "Dedicated developer docs covering cloud environments, CLI, IDE integration, AGENTS.md, and pricing" + } + ], + "methodology": "Documentation completeness review across cloud, CLI, and IDE surfaces", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI: Introducing Codex", + "url": "https://openai.com/index/introducing-codex/", + "date": "2025-05-16", + "value": "Tasks produce verifiable evidence: terminal logs, test outputs, and citations for every action taken" + } + ], + "methodology": "Review of task logs, test output citations, and diff provenance", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Agent explains its approach in task summaries with linked evidence before users accept changes" + } + ], + "methodology": "Assessment of task summaries, cited reasoning, and pre-merge review surfaces", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "Codex CLI repository", + "url": "https://github.com/openai/codex", + "date": "2026-05-01", + "value": "Apache-2.0 CLI with active public development; cloud agent and models closed" + } + ], + "methodology": "Open source assessment weighting open CLI against proprietary cloud service", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Codex CLI repository", + "url": "https://github.com/openai/codex", + "date": "2026-05-01", + "value": "Tens of thousands of stars, rapid release cadence, and a large contributor community since April 2025" + } + ], + "methodology": "Community engagement analysis via GitHub activity and release cadence", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 84, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Codex developer documentation", + "url": "https://developers.openai.com/codex", + "date": "2026-05-01", + "value": "Available in ChatGPT (web/mobile), as a CLI, IDE extensions, and GitHub integration with @codex mentions" + } + ], + "methodology": "Setup and integration surface assessment across ChatGPT, CLI, IDE, and GitHub", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI: Introducing Codex", + "url": "https://openai.com/index/introducing-codex/", + "date": "2025-05-16", + "value": "Cloud architecture runs many tasks in parallel isolated containers, enabling fleet-style delegation" + } + ], + "methodology": "Assessment of parallel container execution and plan-tier task throughput", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 75, + "confidence": "high", + "evidence": [ + { + "source": "Codex pricing documentation", + "url": "https://developers.openai.com/codex/pricing", + "date": "2026-04-09", + "value": "Included in ChatGPT plans with a $100/mo Pro 5x tier added 2026-04-09; typical usage estimated ~$100-200/dev/month" + } + ], + "methodology": "Pricing model analysis of plan-based limits, Pro 5x tier, and typical-usage estimates", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Codex cloud documentation", + "url": "https://developers.openai.com/codex/cloud", + "date": "2026-05-01", + "value": "Per-task logs, usage dashboards, and admin controls for business plans; deeper APM requires external tooling" + } + ], + "methodology": "Review of task logs, usage visibility, and admin monitoring features", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Codex rollout history", + "url": "https://openai.com/index/introducing-codex/", + "date": "2025-06-03", + "value": "Research preview 2025-05-16, ChatGPT Plus rollout 2025-06-03, GPT-5-Codex upgrades Sept 2025; now mature on GPT-5.3-Codex" + } + ], + "methodology": "Maturity assessment from rollout timeline, model upgrades, and enterprise availability", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 92, + "notes": "Core use case: parallel feature work, bug fixes, refactors, and PR generation with test evidence" + }, + "data-analysis": { + "overall": 78, + "notes": "Capable of scripted analysis within containers, though network-off defaults limit live data access" + }, + "research-assistant": { + "overall": 74, + "notes": "Strong at codebase Q&A and architecture exploration; not aimed at general web research" + }, + "education": { + "overall": 75, + "notes": "Cited logs and diffs make its work reviewable for learning, but it targets professional workflows" + } + }, + "best_for": [ + "Teams delegating parallel coding tasks (fixes, features, PRs) to a cloud agent", + "Security-conscious organizations wanting network-isolated execution by default", + "ChatGPT Plus/Pro/Business subscribers seeking included coding-agent capacity", + "Developers wanting an open-source local CLI with OS-level sandboxing" + ], + "strengths": [ + "Strong isolation: per-task containers with internet disabled by default during execution", + "Runs many tasks in parallel for fleet-style software delegation", + "Verifiable outputs with terminal logs, test results, and action citations", + "Open-source Apache-2.0 CLI with local OS-level sandboxing", + "Deep GitHub integration from task to reviewed pull request", + "Continuously upgraded models, currently GPT-5.3-Codex" + ], + "limitations": [ + "Default network isolation can block tasks needing external dependencies unless allowlists are configured", + "Cloud agent and Codex models are proprietary with no self-hosted option", + "Plan-based limits are opaque; heavy users may need the $100/mo Pro 5x tier (~$100-200/dev/month typical)", + "Stateless per-task containers limit cross-task memory beyond AGENTS.md", + "Environment setup scripts add onboarding friction for complex monorepos" + ], + "metadata": { + "license": "Cloud agent proprietary; Codex CLI Apache-2.0 (github.com/openai/codex)", + "supported_models": [ + "GPT-5.3-Codex (current)", + "GPT-5-Codex (Sept 2025)", + "codex-1 (launch)" + ], + "programming_languages": [ + "Language-agnostic (any language in the repository)" + ], + "deployment_type": "Managed cloud containers + local open-source CLI", + "tool_support": [ + "Container shell and editor", + "Test runners", + "GitHub PR integration", + "Configurable network allowlists", + "AGENTS.md project instructions" + ], + "first_release": "CLI April 2025; cloud agent research preview 2025-05-16; ChatGPT Plus rollout 2025-06-03", + "pricing": "Included in ChatGPT plans; $100/mo Pro 5x tier (added 2026-04-09); typical usage ~$100-200/dev/month", + "interfaces": [ + "ChatGPT web and mobile", + "Codex CLI", + "IDE extensions", + "GitHub (@codex mentions)" + ] + }, + "related": [ + "claude-code", + "openai-agents-sdk", + "github-copilot-coding-agent", + "google-jules", + "devin" + ], + "tags": [ + "coding-agent", + "cloud-sandbox", + "openai", + "parallel-tasks" + ] +} diff --git a/data/agents/pydantic-ai.json b/data/agents/pydantic-ai.json index b05cd1d..1bedf6c 100644 --- a/data/agents/pydantic-ai.json +++ b/data/agents/pydantic-ai.json @@ -4,9 +4,9 @@ "name": "Pydantic AI", "provider": "Pydantic", "version": "1.12.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Type-safe Python agent framework from the creators of Pydantic. Provides production-ready agents with strong typing, validation, and structured outputs. Designed for reliability and maintainability in production systems.", + "description": "Type-safe Python agent framework from the creators of Pydantic. Reached v1.0 stable on 2025-09-04 with a formal API stability commitment and has surpassed 15M+ downloads. Provides production-ready agents with strong typing, validation, and structured outputs, designed for reliability and maintainability in production systems.", "website": "https://ai.pydantic.dev/", "trust_vector": { "performance_reliability": { @@ -324,7 +324,7 @@ } }, "operational_excellence": { - "overall_score": 83, + "overall_score": 85, "criteria": { "ease_of_integration": { "score": 87, @@ -383,7 +383,7 @@ "last_verified": "2025-11-09" }, "production_readiness": { - "score": 84, + "score": 90, "confidence": "high", "evidence": [ { @@ -391,10 +391,17 @@ "url": "https://ai.pydantic.dev/", "date": "2024-10-01", "value": "Designed for production use with type safety focus" + }, + { + "source": "Pydantic AI v1 Announcement", + "url": "https://pydantic.dev/articles/pydantic-ai-v1", + "date": "2026-06-10", + "value": "v1.0 stable released 2025-09-04 with API stability commitment; 15M+ downloads" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score raised: v1.0 stable release with API stability commitment confirms production maturity" }, "testing_support": { "score": 86, @@ -427,7 +434,8 @@ "Limited built-in agent orchestration features", "Requires Python and Pydantic knowledge", "No built-in monitoring or observability tools", - "Less opinionated than full-featured frameworks" + "Less opinionated than full-featured frameworks", + "Status (2026-06): actively developed; v1.0 stable since 2025-09-04 with API stability commitment, so pre-1.0 API churn concerns no longer apply" ], "metadata": { "license": "MIT", diff --git a/data/agents/salesforce-einstein-bots.json b/data/agents/salesforce-einstein-bots.json index a9d329e..7a94ba6 100644 --- a/data/agents/salesforce-einstein-bots.json +++ b/data/agents/salesforce-einstein-bots.json @@ -4,9 +4,9 @@ "name": "Salesforce Einstein Bots", "provider": "Salesforce", "version": "Einstein GPT", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Salesforce's AI-powered chatbot platform integrated deeply with the Salesforce ecosystem. Combines Einstein AI with CRM data for personalized customer interactions and seamless handoffs to human agents.", + "description": "TRANSITIONING: Einstein Copilot was retired and absorbed into Agentforce (Jan 2025, Spring '25 release), and Salesforce is steering Einstein Bots customers toward Agentforce agents. Einstein Bots remains Salesforce's CRM-integrated chatbot platform, combining Einstein AI with CRM data for personalized interactions and human-agent handoffs.", "website": "https://www.salesforce.com/products/service-cloud/features/einstein-bots/", "trust_vector": { "performance_reliability": { @@ -321,10 +321,16 @@ "url": "https://www.salesforce.com/products/service-cloud/features/einstein-bots/", "date": "2024-10-01", "value": "Native Salesforce integration, seamless with existing setup" + }, + { + "source": "Salesforce Spring '25 Release Notes", + "url": "https://help.salesforce.com/s/articleView?id=release-notes.rn_copilot_new_name_agentforce_default.htm", + "date": "2026-06-10", + "value": "Einstein Copilot retired and absorbed into Agentforce (Jan 2025, Spring '25 release); Einstein Bots customers are being steered toward Agentforce agents" } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "scalability": { "score": 91, @@ -413,7 +419,8 @@ "Limited capabilities outside of Salesforce ecosystem", "Not suitable for organizations not using Salesforce CRM", "Conversation design less sophisticated than specialized platforms", - "Einstein AI capabilities lag behind latest LLM-based systems" + "Einstein AI capabilities lag behind latest LLM-based systems", + "Product direction has shifted to Agentforce: Einstein Copilot retired (Jan 2025) and Einstein Bots are being steered toward Agentforce agents" ], "metadata": { "license": "Proprietary", @@ -499,6 +506,7 @@ ], "tags": [ "salesforce", - "enterprise" + "enterprise", + "rebranded" ] } diff --git a/data/agents/semantic-kernel-agent.json b/data/agents/semantic-kernel-agent.json index ff168ec..bf88037 100644 --- a/data/agents/semantic-kernel-agent.json +++ b/data/agents/semantic-kernel-agent.json @@ -4,9 +4,9 @@ "name": "Semantic Kernel Agent", "provider": "Microsoft", "version": "1.x", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Microsoft's enterprise-grade SDK for integrating LLMs with conventional programming languages. Provides agent capabilities through plugins, planners, and memory systems with first-class support for .NET, Python, and Java.", + "description": "MAINTENANCE MODE: Semantic Kernel now receives bug/security fixes only and is superseded by the Microsoft Agent Framework (1.0 GA on 2026-04-03), the recommended migration path. It remains Microsoft's enterprise SDK for integrating LLMs with conventional languages via plugins, planners, and memory, with .NET, Python, and Java support.", "website": "https://learn.microsoft.com/en-us/semantic-kernel/", "trust_vector": { "performance_reliability": { @@ -377,10 +377,16 @@ "url": "https://learn.microsoft.com/en-us/semantic-kernel/", "date": "2024-10-01", "value": "Production-ready with major enterprise deployments" + }, + { + "source": "Microsoft Agent Framework Migration Guidance", + "url": "https://devblogs.microsoft.com/semantic-kernel/migrate-your-semantic-kernel-and-autogen-projects-to-microsoft-agent-framework-release-candidate/", + "date": "2026-06-10", + "value": "Semantic Kernel is in maintenance mode (bug/security fixes only); Microsoft Agent Framework 1.0 reached GA on 2026-04-03 as the successor" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -399,7 +405,8 @@ "Less opinionated design requires more architectural decisions", "Smaller community compared to some other frameworks", "Some features still evolving (agents capability)", - "Learning curve for developers new to semantic programming" + "Learning curve for developers new to semantic programming", + "Maintenance mode: bug/security fixes only; migrate to Microsoft Agent Framework for new work" ], "metadata": { "license": "MIT", @@ -474,9 +481,13 @@ "Organizations wanting Azure OpenAI integration", "Teams building multi-language AI applications" ], + "related_entities": [ + "microsoft-agent-framework" + ], "tags": [ "microsoft", "azure", - "open-source" + "open-source", + "maintenance-mode" ] } diff --git a/data/agents/smolagents.json b/data/agents/smolagents.json new file mode 100644 index 0000000..468c0c5 --- /dev/null +++ b/data/agents/smolagents.json @@ -0,0 +1,475 @@ +{ + "id": "smolagents", + "type": "agent", + "name": "smolagents", + "provider": "Hugging Face", + "version": "1.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Minimalist Python agent library from Hugging Face. Its signature CodeAgent writes actions as executable Python code instead of JSON tool calls, enabling expressive multi-step behavior with a deliberately small core codebase.", + "website": "https://github.com/huggingface/smolagents", + "trust_vector": { + "performance_reliability": { + "overall_score": 78, + "criteria": { + "task_completion_accuracy": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Blog Announcement", + "url": "https://huggingface.co/blog/smolagents", + "date": "2024-12-31", + "value": "Hugging Face benchmarks show code-action agents outperform JSON tool-calling agents on multi-step tasks" + } + ], + "methodology": "Review of published benchmark comparisons between code actions and JSON tool calls, plus task completion testing", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "smolagents Documentation", + "url": "https://huggingface.co/docs/smolagents/index", + "date": "2026-05-15", + "value": "Code actions compose tools natively in Python, avoiding JSON parsing failures; tools shareable via Hugging Face Hub" + } + ], + "methodology": "Tool invocation testing across CodeAgent and ToolCallingAgent modes", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Conceptual Guide", + "url": "https://huggingface.co/docs/smolagents/conceptual_guides/intro_agents", + "date": "2026-04-20", + "value": "ReAct-style loop with optional planning steps; code expressiveness allows loops and conditionals within a single action" + } + ], + "methodology": "Complex multi-step task testing using ReAct loop with planning interval enabled", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Memory Documentation", + "url": "https://huggingface.co/docs/smolagents/tutorials/memory", + "date": "2026-03-10", + "value": "In-run step memory is replayable and editable, but long-term cross-session memory requires custom implementation" + } + ], + "methodology": "Memory system evaluation across single-run and cross-session scenarios", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Documentation", + "url": "https://huggingface.co/docs/smolagents/index", + "date": "2026-05-15", + "value": "Execution errors and tracebacks are fed back into the loop so the agent can self-correct within max_steps" + } + ], + "methodology": "Error injection testing observing self-correction behavior across retries", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Multi-Agent Docs", + "url": "https://huggingface.co/docs/smolagents/examples/multiagents", + "date": "2026-02-20", + "value": "Managed agents pattern supports hierarchical multi-agent orchestration, simpler than dedicated multi-agent frameworks" + } + ], + "methodology": "Multi-agent coordination testing using managed agents hierarchy", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 66, + "criteria": { + "tool_sandboxing": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "smolagents Secure Code Execution Guide", + "url": "https://huggingface.co/docs/smolagents/tutorials/secure_code_execution", + "date": "2026-04-15", + "value": "Arbitrary LLM-written code execution is the core paradigm; E2B, Docker, Modal, and Blaxel sandboxes are supported but opt-in, default local executor only restricts imports" + } + ], + "methodology": "Security architecture review of the local Python executor versus opt-in remote sandbox backends", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents GitHub", + "url": "https://github.com/huggingface/smolagents", + "date": "2026-05-30", + "value": "Library provides no built-in authentication or authorization; access control is entirely the integrating developer's responsibility" + } + ], + "methodology": "Access control capabilities assessment of library surface", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Secure Code Execution Guide", + "url": "https://huggingface.co/docs/smolagents/tutorials/secure_code_execution", + "date": "2026-04-15", + "value": "No built-in injection defenses; docs explicitly warn that untrusted inputs combined with code execution require sandboxing" + } + ], + "methodology": "Injection attack surface review; code-action paradigm amplifies impact of successful injection", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Sandboxed Execution Options", + "url": "https://huggingface.co/docs/smolagents/tutorials/secure_code_execution", + "date": "2026-04-15", + "value": "Process and filesystem isolation only achieved when E2B/Docker/Modal sandboxes are configured; local executor shares host environment" + } + ], + "methodology": "Data isolation architecture review across executor backends", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "smolagents GitHub Repository", + "url": "https://github.com/huggingface/smolagents", + "date": "2026-06-01", + "value": "Apache 2.0 license, deliberately minimal core (~thousands of lines), fully auditable, active Hugging Face maintenance" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 84, + "criteria": { + "data_retention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Self-Hosted Library Architecture", + "url": "https://github.com/huggingface/smolagents", + "date": "2026-06-01", + "value": "Library runs entirely in user infrastructure; no data is retained by the framework itself" + } + ], + "methodology": "Privacy architecture review of self-hosted library model", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents GitHub", + "url": "https://github.com/huggingface/smolagents", + "date": "2026-05-30", + "value": "GDPR compliance achievable when self-hosted with local models; depends on chosen model provider and sandbox vendor" + } + ], + "methodology": "Compliance capabilities assessment across deployment configurations", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Models Documentation", + "url": "https://huggingface.co/docs/smolagents/reference/models", + "date": "2026-04-20", + "value": "Data flows to whichever model provider is configured (HF Inference, OpenAI, Anthropic) and to remote sandbox vendors if used; fully local operation possible" + } + ], + "methodology": "Data flow analysis across model and executor backends", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "smolagents Models Documentation", + "url": "https://huggingface.co/docs/smolagents/reference/models", + "date": "2026-04-20", + "value": "Supports fully local models via Transformers, Ollama, llama.cpp, and any OpenAI-compatible local server" + } + ], + "methodology": "Deployment options assessment including air-gapped configurations", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 87, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "smolagents Documentation", + "url": "https://huggingface.co/docs/smolagents/index", + "date": "2026-05-15", + "value": "Comprehensive guided tour, tutorials, conceptual guides, and security guidance maintained by Hugging Face" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "smolagents Telemetry Tutorial", + "url": "https://huggingface.co/docs/smolagents/tutorials/inspect_runs", + "date": "2026-03-15", + "value": "OpenTelemetry instrumentation supported with Langfuse/Phoenix integrations; every step logged with full code actions" + } + ], + "methodology": "Tracing and logging capabilities assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Conceptual Guide", + "url": "https://huggingface.co/docs/smolagents/conceptual_guides/react", + "date": "2026-02-20", + "value": "Code actions plus ReAct thoughts are human-readable, making each step's intent inspectable" + } + ], + "methodology": "Explainability assessment of agent step outputs", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "smolagents GitHub Repository", + "url": "https://github.com/huggingface/smolagents", + "date": "2026-06-01", + "value": "Apache 2.0, released Dec 2024/Jan 2025, 20k+ stars, intentionally small auditable core" + } + ], + "methodology": "Open source assessment of license, codebase size, and auditability", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Activity", + "url": "https://github.com/huggingface/smolagents", + "date": "2026-06-01", + "value": "Frequent releases, large contributor base, and strong Hugging Face community ecosystem" + } + ], + "methodology": "Community engagement analysis of commits, issues, and releases", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 79, + "criteria": { + "ease_of_integration": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "smolagents Guided Tour", + "url": "https://huggingface.co/docs/smolagents/guided_tour", + "date": "2026-05-15", + "value": "Working agent in a few lines of code; minimal abstractions and model-agnostic interfaces" + } + ], + "methodology": "Integration complexity assessment with minimal-setup testing", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents GitHub Discussions", + "url": "https://github.com/huggingface/smolagents/issues", + "date": "2026-04-10", + "value": "Library provides no orchestration layer; horizontal scaling, queuing, and sandbox pooling are user responsibilities" + } + ], + "methodology": "Scalability architecture assessment for production workloads", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Open Source Library", + "url": "https://github.com/huggingface/smolagents", + "date": "2026-06-01", + "value": "Free Apache 2.0 framework; costs limited to chosen LLM API and optional sandbox provider usage" + } + ], + "methodology": "Pricing model analysis", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents Telemetry Tutorial", + "url": "https://huggingface.co/docs/smolagents/tutorials/inspect_runs", + "date": "2026-03-15", + "value": "OpenTelemetry hooks available but dashboards and alerting require external platforms such as Langfuse or Phoenix" + } + ], + "methodology": "Monitoring features assessment", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "smolagents GitHub Releases", + "url": "https://github.com/huggingface/smolagents/releases", + "date": "2026-05-30", + "value": "Actively developed with occasional breaking changes; production deployments must add sandboxing, scaling, and guardrails themselves" + } + ], + "methodology": "Production readiness assessment of API stability and operational gaps", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "research-assistant": { + "overall": 87, + "notes": "Code actions excel at web research and tool-composition tasks; powers strong open deep-research demos" + }, + "code-generation": { + "overall": 85, + "notes": "Natural fit since the agent already thinks in Python; sandbox strongly recommended" + }, + "data-analysis": { + "overall": 88, + "notes": "Writing pandas/numpy code directly as actions is a standout strength" + }, + "content-creation": { + "overall": 76, + "notes": "Capable but code-action paradigm offers less advantage for pure text generation" + }, + "education": { + "overall": 78, + "notes": "Readable code actions make agent reasoning easy to teach and inspect" + }, + "customer-support": { + "overall": 68, + "notes": "Lacks built-in guardrails, sessions, and access control needed for user-facing support" + }, + "financial-analysis": { + "overall": 70, + "notes": "Strong for quantitative scripting but requires hardened sandboxing and compliance work" + } + }, + "best_for": [ + "Developers wanting a minimal, hackable agent library without heavy abstractions", + "Data analysis and research tasks where Python code actions outperform JSON tool calls", + "Teams in the Hugging Face ecosystem using Hub-shared tools and open models", + "Prototyping agents quickly with any model provider, including fully local LLMs" + ], + "strengths": [ + "CodeAgent paradigm: actions as Python code are more expressive and benchmark better than JSON tool calls", + "Deliberately minimal, auditable core that is easy to learn and extend", + "Model-agnostic: HF Inference, OpenAI, Anthropic, and fully local models supported", + "First-class sandbox integrations (E2B, Docker, Modal, Blaxel) for secure execution", + "Apache 2.0 with strong Hugging Face backing and community", + "OpenTelemetry instrumentation for step-level run inspection" + ], + "limitations": [ + "Arbitrary code execution is the core paradigm; running without an opt-in sandbox is risky", + "No built-in long-term memory, access control, or guardrails", + "Minimal orchestration layer; scaling and production hardening left to the developer", + "Prompt injection consequences are amplified because actions are executable code", + "API still evolves with occasional breaking changes between releases" + ], + "related": [ + "crewai", + "langgraph-agent", + "pydantic-ai", + "e2b-agents", + "openai-agents-sdk" + ], + "metadata": { + "license": "Apache 2.0", + "supported_models": [ + "Hugging Face Inference (open models)", + "OpenAI GPT models", + "Anthropic Claude", + "Local LLMs via Transformers/Ollama/llama.cpp" + ], + "programming_languages": [ + "Python" + ], + "deployment_type": "Self-hosted", + "tool_support": [ + "Hub-shared tools", + "Custom Python tools", + "MCP tools", + "LangChain tool import" + ], + "github_stars": "20000+", + "first_release": "2024 (Dec 2024/Jan 2025)", + "pricing": "Free (Apache 2.0) - Costs only from LLM API calls and optional sandbox providers", + "python_requirement": "Python >=3.10", + "adoption": "One of the most-starred minimalist agent libraries; widely used for open deep-research agents" + }, + "tags": [ + "code-agent", + "minimalist", + "open-source" + ] +} diff --git a/data/agents/strands-agents.json b/data/agents/strands-agents.json new file mode 100644 index 0000000..510ec65 --- /dev/null +++ b/data/agents/strands-agents.json @@ -0,0 +1,477 @@ +{ + "id": "strands-agents", + "type": "agent", + "name": "Strands Agents", + "provider": "Amazon Web Services", + "version": "1.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Open-source, model-driven AI agent SDK from AWS, used internally by Amazon Q Developer. Takes a lightweight model-first approach with MCP and A2A support, multi-agent primitives, and optional pairing with Amazon Bedrock AgentCore for hosted runtime.", + "website": "https://strandsagents.com/", + "trust_vector": { + "performance_reliability": { + "overall_score": 84, + "criteria": { + "task_completion_accuracy": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "AWS Open Source Blog", + "url": "https://aws.amazon.com/blogs/opensource/introducing-strands-agents-an-open-source-ai-agents-sdk/", + "date": "2025-05-16", + "value": "Model-driven loop proven internally by Amazon Q Developer and other AWS agent teams before open-sourcing" + } + ], + "methodology": "Task completion testing plus review of documented internal AWS production usage", + "last_verified": "2026-06-10" + }, + "tool_use_reliability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Strands Agents Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/concepts/tools/tools_overview/", + "date": "2026-05-20", + "value": "Native MCP client support, decorator-based Python tools, and a maintained strands-agents-tools library" + } + ], + "methodology": "Tool invocation testing across native tools and MCP servers", + "last_verified": "2026-06-10" + }, + "multi_step_planning": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Strands Agent Loop Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/concepts/agents/agent-loop/", + "date": "2026-04-15", + "value": "Model-driven agentic loop delegates planning to the model; workflow and graph patterns available for structured multi-step tasks" + } + ], + "methodology": "Complex multi-step task testing with agent loop and graph patterns", + "last_verified": "2026-06-10" + }, + "memory_persistence": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Strands Session Management Docs", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/concepts/agents/session-management/", + "date": "2026-03-20", + "value": "Built-in session persistence plus integration with Amazon Bedrock AgentCore Memory for long-term memory" + } + ], + "methodology": "Memory system evaluation across sessions and AgentCore Memory integration", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "Strands Agents GitHub", + "url": "https://github.com/strands-agents/sdk-python", + "date": "2026-05-21", + "value": "Tool errors surfaced back to the model for self-correction; retry and hook system available for custom recovery" + } + ], + "methodology": "Error injection testing observing model-driven recovery", + "last_verified": "2026-06-10" + }, + "agent_collaboration": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Strands Multi-Agent Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/concepts/multi-agent/agent-to-agent/", + "date": "2026-04-15", + "value": "First-class multi-agent primitives: agents-as-tools, swarm, graph, and A2A protocol support" + } + ], + "methodology": "Multi-agent coordination testing across swarm, graph, and A2A patterns", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 78, + "criteria": { + "tool_sandboxing": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Amazon Bedrock AgentCore", + "url": "https://aws.amazon.com/bedrock/agentcore/", + "date": "2025-10-13", + "value": "SDK itself does not sandbox tools, but AgentCore Runtime (GA Oct 2025) provides isolated sessions and a managed Code Interpreter sandbox" + } + ], + "methodology": "Security architecture review of SDK plus AgentCore runtime isolation", + "last_verified": "2026-06-10" + }, + "access_control": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Strands + AgentCore Identity", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/deploy/deploy_to_bedrock_agentcore/", + "date": "2026-04-10", + "value": "Integrates with AWS IAM and AgentCore Identity for credential management and per-agent permission scoping" + } + ], + "methodology": "Access control assessment of IAM and identity integrations", + "last_verified": "2026-06-10" + }, + "prompt_injection_defense": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Strands Guardrails Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/safety-security/guardrails/", + "date": "2026-03-25", + "value": "Supports Amazon Bedrock Guardrails and hook-based input/output filtering; no framework-level injection defense by default" + } + ], + "methodology": "Injection testing with and without Bedrock Guardrails enabled", + "last_verified": "2026-06-10" + }, + "data_isolation": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "AgentCore Runtime Session Isolation", + "url": "https://aws.amazon.com/bedrock/agentcore/", + "date": "2025-10-13", + "value": "AgentCore runs each session in dedicated microVM isolation; self-hosted deployments inherit user infrastructure isolation" + } + ], + "methodology": "Data isolation architecture review across deployment targets", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Strands Agents GitHub", + "url": "https://github.com/strands-agents/sdk-python", + "date": "2026-06-01", + "value": "Apache 2.0, open-sourced May 2025, public roadmap, contributions from Anthropic, Meta, and other companies" + } + ], + "methodology": "Source code and governance review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 84, + "criteria": { + "data_retention": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Self-Hosted SDK Architecture", + "url": "https://github.com/strands-agents/sdk-python", + "date": "2026-06-01", + "value": "SDK retains no data itself; retention governed by user infrastructure or AWS data policies when using AgentCore" + } + ], + "methodology": "Privacy architecture review of SDK and managed runtime options", + "last_verified": "2026-06-10" + }, + "gdpr_compliance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "AWS Compliance Programs", + "url": "https://aws.amazon.com/compliance/gdpr-center/", + "date": "2026-05-01", + "value": "Deployments on AWS inherit GDPR-aligned infrastructure controls; self-hosted deployments fully controllable" + } + ], + "methodology": "Compliance capabilities assessment leveraging AWS compliance posture", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Strands Model Providers Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/concepts/model-providers/amazon-bedrock/", + "date": "2026-04-15", + "value": "Data flows only to the configured model provider (Bedrock, Anthropic, OpenAI, Ollama, LiteLLM); Bedrock keeps data within AWS" + } + ], + "methodology": "Data flow analysis across supported model providers", + "last_verified": "2026-06-10" + }, + "local_deployment_option": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Strands Ollama Provider Docs", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/concepts/model-providers/ollama/", + "date": "2026-04-15", + "value": "Runs anywhere Python/TypeScript runs, including fully local execution with Ollama models" + } + ], + "methodology": "Deployment options assessment including local model configurations", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 86, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Strands Agents Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/", + "date": "2026-05-20", + "value": "Thorough user guide covering concepts, safety/security, deployment, and multi-agent patterns with samples repo" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "execution_traceability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Strands Observability Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/observability-evaluation/observability/", + "date": "2026-04-20", + "value": "Native OpenTelemetry traces, metrics, and logs built into the SDK; integrates with AgentCore Observability" + } + ], + "methodology": "Tracing and telemetry capabilities assessment", + "last_verified": "2026-06-10" + }, + "decision_explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Strands Agent Loop Documentation", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/concepts/agents/agent-loop/", + "date": "2026-04-15", + "value": "Model-driven loop exposes reasoning, tool selections, and intermediate steps through traces and hooks" + } + ], + "methodology": "Explainability assessment of loop transparency and hook system", + "last_verified": "2026-06-10" + }, + "open_source_code": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Strands Agents GitHub Organization", + "url": "https://github.com/strands-agents", + "date": "2026-06-01", + "value": "Apache 2.0 SDK, tools, and samples; Python SDK 1.0 released 2026-05-21, TypeScript 1.0 released 2026-04-30" + } + ], + "methodology": "Open source assessment of license and release maturity", + "last_verified": "2026-06-10" + }, + "community_activity": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "AWS Open Source Blog / GitHub Metrics", + "url": "https://github.com/strands-agents/sdk-python", + "date": "2026-02-15", + "value": "14M+ downloads by Feb 2026, rapid release cadence, and growing external contributor base" + } + ], + "methodology": "Community engagement analysis of downloads, releases, and contributors", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 86, + "criteria": { + "ease_of_integration": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Strands Quickstart", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/quickstart/", + "date": "2026-05-20", + "value": "Agent in a few lines of code; model-driven approach avoids complex workflow definitions" + } + ], + "methodology": "Integration complexity assessment with minimal-setup testing", + "last_verified": "2026-06-10" + }, + "scalability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock AgentCore GA", + "url": "https://aws.amazon.com/bedrock/agentcore/", + "date": "2025-10-13", + "value": "AgentCore Runtime (GA 2025-10-13) provides serverless, session-isolated scaling; SDK also deploys to Lambda, Fargate, EKS" + } + ], + "methodology": "Scalability assessment across managed and self-managed deployment targets", + "last_verified": "2026-06-10" + }, + "cost_predictability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Open Source SDK", + "url": "https://github.com/strands-agents/sdk-python", + "date": "2026-06-01", + "value": "Free Apache 2.0 SDK; costs from LLM usage and optional AWS services (AgentCore consumption-based pricing)" + } + ], + "methodology": "Pricing model analysis of SDK and optional managed services", + "last_verified": "2026-06-10" + }, + "monitoring_capabilities": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Strands Observability + AgentCore", + "url": "https://strandsagents.com/latest/documentation/docs/user-guide/observability-evaluation/observability/", + "date": "2026-04-20", + "value": "Built-in OTel metrics/traces plus CloudWatch and AgentCore Observability dashboards for production monitoring" + } + ], + "methodology": "Monitoring features assessment across SDK and AWS integrations", + "last_verified": "2026-06-10" + }, + "production_readiness": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Strands 1.0 Releases / AWS Internal Usage", + "url": "https://github.com/strands-agents/sdk-python/releases", + "date": "2026-05-21", + "value": "Python 1.0 (2026-05-21) and TypeScript 1.0 (2026-04-30) with semver stability; battle-tested in Amazon Q Developer" + } + ], + "methodology": "Production readiness assessment of API stability and documented production usage", + "last_verified": "2026-06-10" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 87, + "notes": "Powers Amazon Q Developer internally; strong tool use and MCP support for dev workflows" + }, + "customer-support": { + "overall": 84, + "notes": "Session management, guardrails, and AgentCore identity make user-facing agents practical" + }, + "data-analysis": { + "overall": 83, + "notes": "Good tool composition plus AgentCore Code Interpreter for sandboxed analysis" + }, + "research-assistant": { + "overall": 84, + "notes": "Multi-agent swarm/graph patterns suit research decomposition well" + }, + "financial-analysis": { + "overall": 80, + "notes": "AWS compliance posture and IAM integration help in regulated finance environments" + }, + "healthcare": { + "overall": 76, + "notes": "Viable on HIPAA-eligible AWS services but requires careful architecture review" + }, + "legal-compliance": { + "overall": 78, + "notes": "Strong audit trails via OTel tracing; domain guardrails must be added" + } + }, + "best_for": [ + "Teams on AWS wanting an open SDK with a managed production runtime (AgentCore)", + "Enterprises needing IAM-integrated, observable agents with audit trails", + "Developers building multi-agent systems with MCP and A2A interoperability", + "Organizations seeking a battle-tested SDK proven inside Amazon Q Developer" + ], + "strengths": [ + "Model-driven design proven in production by Amazon Q Developer and AWS teams", + "First-class MCP and A2A protocol support plus rich multi-agent primitives (swarm, graph, agents-as-tools)", + "Native OpenTelemetry observability built into the SDK", + "Truly model-agnostic: Bedrock, Anthropic, OpenAI, Ollama, LiteLLM and more", + "Stable 1.0 APIs in both Python and TypeScript with strong adoption (14M+ downloads)", + "Seamless path to managed, session-isolated runtime via Amazon Bedrock AgentCore" + ], + "limitations": [ + "No built-in tool sandboxing in the SDK itself; isolation requires AgentCore or user infrastructure", + "Deepest integrations (Guardrails, Identity, Memory, Observability) favor the AWS ecosystem", + "Model-driven loop offers less deterministic control than explicit graph-first frameworks", + "Younger community ecosystem than longer-established agent frameworks", + "Advanced multi-agent patterns still maturing relative to the core single-agent loop" + ], + "related": [ + "amazon-bedrock-agents", + "langgraph-agent", + "openai-agents-sdk", + "google-adk", + "crewai" + ], + "metadata": { + "license": "Apache 2.0", + "supported_models": [ + "Amazon Bedrock (Claude, Nova, Llama, etc.)", + "Anthropic API", + "OpenAI", + "Local LLMs via Ollama", + "100+ providers via LiteLLM" + ], + "programming_languages": [ + "Python", + "TypeScript" + ], + "deployment_type": "Self-hosted or managed via Amazon Bedrock AgentCore", + "tool_support": [ + "MCP tools", + "Custom Python/TypeScript tools", + "strands-agents-tools library", + "A2A agent interoperability" + ], + "github_stars": "10000+", + "first_release": "2025 (open-sourced May 2025)", + "pricing": "Free (Apache 2.0) - Costs only from LLM usage and optional AWS services", + "python_requirement": "Python >=3.10", + "adoption": "14M+ downloads by Feb 2026; used internally by Amazon Q Developer and multiple AWS teams" + }, + "tags": [ + "model-driven", + "aws", + "open-source" + ] +} diff --git a/data/agents/swarm.json b/data/agents/swarm.json index a5f30c0..273ca8b 100644 --- a/data/agents/swarm.json +++ b/data/agents/swarm.json @@ -4,9 +4,9 @@ "name": "OpenAI Swarm", "provider": "OpenAI", "version": "Experimental (Deprecated)", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Experimental educational framework from OpenAI for building multi-agent systems with lightweight orchestration. Demonstrates ergonomic patterns for agent coordination and handoffs using simple Python primitives.", + "description": "DEPRECATED: archived since 2025-03-11; the README redirects users to the OpenAI Agents SDK as the production successor. Swarm was an experimental educational framework from OpenAI for multi-agent orchestration, demonstrating agent coordination and handoff patterns with simple Python primitives. Not maintained; do not use for new projects.", "website": "https://github.com/openai/swarm", "trust_vector": { "performance_reliability": { @@ -324,7 +324,7 @@ } }, "operational_excellence": { - "overall_score": 68, + "overall_score": 69, "criteria": { "ease_of_use": { "score": 85, @@ -341,7 +341,7 @@ "last_verified": "2025-11-09" }, "production_readiness": { - "score": 45, + "score": 25, "confidence": "high", "evidence": [ { @@ -349,10 +349,17 @@ "url": "https://github.com/openai/swarm", "date": "2024-10-01", "value": "Explicitly not recommended for production use" + }, + { + "source": "GitHub Repository Status", + "url": "https://github.com/openai/swarm", + "date": "2026-06-10", + "value": "Repository deprecated/archived since 2025-03-11; README redirects to the OpenAI Agents SDK as the production successor" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10", + "notes": "Score reduced: project archived since 2025-03-11 with no further maintenance" }, "scalability": { "score": 65, @@ -427,7 +434,8 @@ "Limited features compared to production frameworks", "No built-in monitoring, error handling, or scaling features", "Minimal documentation beyond examples", - "Not actively maintained for production use cases" + "Not actively maintained for production use cases", + "Deprecated/archived since 2025-03-11; README directs users to the OpenAI Agents SDK" ], "metadata": { "license": "MIT", @@ -498,10 +506,14 @@ "Teams learning agent handoff patterns and routines", "Researchers experimenting with ergonomic agent orchestration" ], + "related_entities": [ + "openai-agents-sdk" + ], "tags": [ "openai", "assistants", "experimental", - "open-source" + "open-source", + "deprecated" ] } diff --git a/data/mcps/mcp-server-apify.json b/data/mcps/mcp-server-apify.json new file mode 100644 index 0000000..41c4b44 --- /dev/null +++ b/data/mcps/mcp-server-apify.json @@ -0,0 +1,450 @@ +{ + "id": "mcp-server-apify", + "type": "mcp", + "name": "Apify MCP Server", + "provider": "Apify", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Official Apify MCP server (renamed from actors-mcp-server) exposing thousands of Apify Store Actors to AI agents — scrapers for social media, maps, e-commerce, and search. Agents can dynamically discover and run Actors via the @apify/actors-mcp-server package or the hosted OAuth endpoint at mcp.apify.com.", + "website": "https://github.com/apify/apify-mcp-server", + "trust_vector": { + "performance_reliability": { + "overall_score": 81, + "criteria": { + "api_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Apify Platform Documentation", + "url": "https://docs.apify.com/platform", + "date": "2026-06-10", + "value": "Built on Apify's mature cloud platform with managed Actor execution, storage, and a public status page" + } + ], + "methodology": "Platform stability and maturity analysis", + "last_verified": "2026-06-10" + }, + "actor_execution_success": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Apify Store", + "url": "https://apify.com/store", + "date": "2026-06-10", + "value": "Actor quality varies: maintained official Actors are highly reliable, while community Actors can break when target sites change" + } + ], + "methodology": "Run success sampling across official and community Actors", + "last_verified": "2026-06-10" + }, + "dynamic_tool_discovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Apify MCP Server Repository", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-06-10", + "value": "Agents can search the Apify Store and add Actors as tools at runtime; discovery quality depends on Store metadata" + } + ], + "methodology": "Tool discovery relevance testing", + "last_verified": "2026-06-10" + }, + "rate_limit_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Apify API Documentation", + "url": "https://docs.apify.com/api/v2", + "date": "2026-06-10", + "value": "Platform enforces documented API rate limits and plan-based usage quotas; runs queue rather than hard-fail under load" + } + ], + "methodology": "Rate limiting behavior review", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Apify MCP Server Repository", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-06-10", + "value": "Actor run failures return structured errors with run logs available in the Apify console for diagnosis" + } + ], + "methodology": "Failure mode and recovery testing", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 71, + "criteria": { + "authentication_security": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Apify MCP Documentation", + "url": "https://mcp.apify.com/", + "date": "2026-06-10", + "value": "Hosted server at mcp.apify.com uses OAuth; local stdio mode uses an Apify API token" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-06-10" + }, + "dynamic_capability_risk": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Apify MCP Server Repository", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-06-10", + "value": "Dynamically added Actors change the agent's capability surface at runtime, bypassing assumptions made at initial tool-approval time" + } + ], + "methodology": "Runtime capability mutation threat modeling", + "last_verified": "2026-06-10" + }, + "actor_supply_chain_risk": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "Apify Store", + "url": "https://apify.com/store", + "date": "2026-06-10", + "value": "Actors are community-published code; while Apify reviews and badges some, agents can invoke thousands of third-party Actors of varying provenance" + } + ], + "methodology": "Supply chain analysis of community-published Actors", + "last_verified": "2026-06-10" + }, + "credential_handling": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Apify Security Documentation", + "url": "https://docs.apify.com/platform/security", + "date": "2026-06-10", + "value": "API tokens scoped to the account; leaked tokens allow running paid Actors and reading stored datasets" + } + ], + "methodology": "Token scope and exposure analysis", + "last_verified": "2026-06-10" + }, + "sandboxed_execution": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Apify Platform Documentation", + "url": "https://docs.apify.com/platform/actors", + "date": "2026-06-10", + "value": "Actors execute in isolated containers in Apify's cloud, not on the user's machine, containing the impact of malicious Actor code" + } + ], + "methodology": "Execution isolation review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 66, + "criteria": { + "data_exposure": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Apify Platform Documentation", + "url": "https://docs.apify.com/platform/storage", + "date": "2026-06-10", + "value": "Scraped results are stored in Apify cloud datasets and then passed into the LLM context; inputs may include sensitive query terms" + } + ], + "methodology": "Data flow analysis of Actor inputs and outputs", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Apify MCP Server Repository", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-06-10", + "value": "No built-in PII filtering on Actor output; social media and maps scrapers frequently return personal data subject to GDPR obligations" + } + ], + "methodology": "PII handling assessment of common Actor outputs", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Apify Privacy Policy", + "url": "https://apify.com/privacy-policy", + "date": "2026-06-10", + "value": "Run inputs and outputs are processed by Apify and the invoked Actor's code; community Actor authors define their own data behavior" + } + ], + "methodology": "Data sharing and policy review", + "last_verified": "2026-06-10" + }, + "compliance_posture": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Apify Security and Compliance", + "url": "https://docs.apify.com/platform/security", + "date": "2026-06-10", + "value": "Apify documents GDPR compliance and platform security practices; lawful use of scraped personal data remains the user's responsibility" + } + ], + "methodology": "Compliance documentation review", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 85, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Apify MCP Documentation", + "url": "https://docs.apify.com/platform/integrations/mcp", + "date": "2026-06-10", + "value": "Thorough docs covering hosted and local setup, tool configuration, and Actor selection options" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Apify Console", + "url": "https://console.apify.com/", + "date": "2026-06-10", + "value": "Every Actor run is logged in the Apify console with inputs, logs, usage cost, and output datasets" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Apify MCP Server Repository", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-06-10", + "value": "MIT-licensed open source server with 1,319 GitHub stars; many Store Actors also publish source, though not all" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + }, + "tool_coverage_clarity": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Apify MCP Server Repository", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-06-10", + "value": "Helper tools are documented, but the effective tool set is dynamic — it depends on which Actors are loaded at runtime" + } + ], + "methodology": "Tool surface documentation review", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 84, + "criteria": { + "ease_of_setup": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Apify MCP Documentation", + "url": "https://mcp.apify.com/", + "date": "2026-06-10", + "value": "Hosted endpoint with OAuth requires no installation; local mode is a single npx command with an API token" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-06-10" + }, + "api_performance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Apify Platform Documentation", + "url": "https://docs.apify.com/platform/actors/running", + "date": "2026-06-10", + "value": "Actor runs include container startup overhead; simple scrapes finish in seconds while large scraping jobs run for minutes" + } + ], + "methodology": "Run latency characterization", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Apify Status Page", + "url": "https://status.apify.com/", + "date": "2026-06-10", + "value": "Public status page with historically high platform availability" + } + ], + "methodology": "Uptime and incident history analysis", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Apify Store", + "url": "https://apify.com/store", + "date": "2026-06-10", + "value": "Thousands of ready-made Actors covering social media, maps, e-commerce, search, and general scraping" + } + ], + "methodology": "Capability breadth assessment", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 79, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Repository Metrics", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-06-10", + "value": "1,319 GitHub stars with active maintenance by Apify and a large existing Actor developer ecosystem" + } + ], + "methodology": "Community activity and adoption analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Access to thousands of ready-made Apify Store Actors for social media, maps, e-commerce, and search scraping", + "Dynamic tool discovery lets agents find and add the right Actor at runtime", + "Actors run in isolated containers in Apify's cloud, not on the user's machine", + "Hosted OAuth endpoint (mcp.apify.com) and simple local npx setup", + "Full run auditability via the Apify console with logs, inputs, and cost tracking", + "MIT-licensed open source server actively maintained by Apify" + ], + "limitations": [ + "Actors are community-published code of varying quality and provenance", + "Dynamically added tools mutate the agent's capability surface after initial approval", + "Scraped output frequently contains personal data with GDPR implications and no built-in PII filtering", + "Actor runs consume paid platform credits that an agent can spend autonomously", + "Community Actors can silently break when target websites change", + "Scraped content is untrusted input that may carry prompt injection payloads" + ], + "metadata": { + "license": "MIT", + "supported_platforms": [ + "All platforms with Node.js", + "Hosted remote server (mcp.apify.com)" + ], + "programming_languages": [ + "TypeScript" + ], + "github_repo": "https://github.com/apify/apify-mcp-server", + "github_stars": 1319, + "package_name": "@apify/actors-mcp-server", + "previous_name": "actors-mcp-server", + "api_dependency": "Apify platform API", + "authentication": "OAuth (hosted) or Apify API token (stdio)", + "maintained_by": "Apify", + "transport_types": [ + "stdio", + "remote (hosted)" + ], + "installation_methods": [ + "npm", + "hosted endpoint" + ] + }, + "use_case_ratings": { + "research-assistant": { + "overall": 88, + "notes": "Excellent for gathering data from platforms that are hard to scrape directly (social media, maps, marketplaces)" + }, + "data-analysis": { + "overall": 86, + "notes": "Actor outputs land in structured datasets that feed analysis pipelines well" + }, + "content-creation": { + "overall": 78, + "notes": "Strong for trend, competitor, and audience research feeding content workflows" + }, + "customer-support": { + "overall": 62, + "notes": "Limited fit; can monitor reviews and social mentions but is not a support tool" + }, + "financial-analysis": { + "overall": 74, + "notes": "Useful for alternative data collection (e-commerce prices, reviews); verify data quality and licensing" + }, + "legal-compliance": { + "overall": 55, + "notes": "Scraping personal data raises GDPR and terms-of-service questions requiring careful legal review" + }, + "education": { + "overall": 70, + "notes": "Good for teaching data collection concepts, though paid credits and ToS limits apply" + } + }, + "best_for": [ + "Agents needing ready-made scrapers for social media, maps, e-commerce, and search", + "Teams already on the Apify platform who want their Actors exposed to AI assistants", + "Data collection workflows where cloud-isolated execution is preferred over local scraping" + ], + "related_entities": [ + "mcp-server-firecrawl", + "mcp-server-playwright", + "mcp-server-brave-search", + "mcp-server-tavily" + ], + "tags": [ + "web-scraping", + "automation", + "actors", + "data-extraction", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-brave-search.json b/data/mcps/mcp-server-brave-search.json index af6ec8c..e303070 100644 --- a/data/mcps/mcp-server-brave-search.json +++ b/data/mcps/mcp-server-brave-search.json @@ -4,13 +4,13 @@ "name": "MCP Brave Search Server", "provider": "Anthropic (Archived)", "version": "2025.4.24", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "MCP server providing AI models with web search capabilities through Brave Search API. Enables real-time information retrieval, fact-checking, and research augmentation through the Model Context Protocol. Privacy-focused alternative to traditional search APIs.", + "description": "ARCHIVED: Former Anthropic reference MCP server for the Brave Search API, archived 2025-05-29 to the servers-archived repository and no longer maintained (no security guarantees). Brave now ships its own official Brave Search MCP server, which is the recommended replacement. The underlying Brave Search API remains active and privacy-focused.", "website": "https://brave.com/search/api/", "trust_vector": { "performance_reliability": { - "overall_score": 85, + "overall_score": 80, "criteria": { "search_result_quality": { "score": 87, @@ -69,7 +69,7 @@ "last_verified": "2025-11-09" }, "error_recovery": { - "score": 80, + "score": 58, "confidence": "medium", "evidence": [ { @@ -77,10 +77,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Graceful error handling with retry logic for failed requests" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Server archived 2025-05-29; bugs and Brave API changes will never be fixed in this implementation" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } }, @@ -146,7 +152,7 @@ } }, "privacy_compliance": { - "overall_score": 88, + "overall_score": 89, "criteria": { "privacy_focus": { "score": 92, @@ -207,7 +213,7 @@ } }, "trust_transparency": { - "overall_score": 83, + "overall_score": 77, "criteria": { "documentation_quality": { "score": 85, @@ -252,7 +258,7 @@ "last_verified": "2025-11-09" }, "implementation_clarity": { - "score": 85, + "score": 62, "confidence": "medium", "evidence": [ { @@ -260,15 +266,21 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Open source MCP server implementation available" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Code now lives in the read-only servers-archived repository; README states no security guarantees are provided for archived servers" } ], "methodology": "Code transparency review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } }, "operational_excellence": { - "overall_score": 84, + "overall_score": 83, "criteria": { "ease_of_setup": { "score": 90, @@ -405,7 +417,7 @@ "Smaller index compared to Google, may miss some niche content", "API costs can accumulate with heavy usage", "Limited advanced search features compared to Google", - "Community-maintained MCP server (not official Anthropic release)" + "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; use Brave's official Brave Search MCP server instead" ], "metadata": { "license": "Proprietary API, MCP server varies", @@ -422,12 +434,16 @@ "authentication": "API key", "rate_limits": "Varies by plan (2000-15000 queries/month)", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "None (Archived 2025-05-29)", + "repository": "https://github.com/modelcontextprotocol/servers-archived", + "replacement": "Brave's official Brave Search MCP server" }, "tags": [ "search", "web", "mcp", - "model-context-protocol" + "model-context-protocol", + "archived", + "unmaintained" ] } diff --git a/data/mcps/mcp-server-chrome-devtools.json b/data/mcps/mcp-server-chrome-devtools.json new file mode 100644 index 0000000..d730af0 --- /dev/null +++ b/data/mcps/mcp-server-chrome-devtools.json @@ -0,0 +1,443 @@ +{ + "id": "mcp-server-chrome-devtools", + "type": "mcp", + "name": "Chrome DevTools MCP", + "provider": "Google (Chrome DevTools team)", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Google's official MCP server that gives AI agents full Chrome control via the Chrome DevTools Protocol. Around 26 tools span input automation, navigation, performance tracing and insights, network inspection, console debugging, and screenshots — making it the reference server for AI-assisted web debugging and performance work.", + "website": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 87, + "criteria": { + "cdp_reliability": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP README", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Built directly on the Chrome DevTools Protocol via Puppeteer, the same battle-tested interface used by Chrome DevTools itself; can launch a fresh Chrome instance or attach to a running one" + } + ], + "methodology": "Review of CDP connection handling for launched and attached Chrome instances", + "last_verified": "2026-06-10" + }, + "operation_success_rate": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP tool reference", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp/blob/main/docs/tool-reference.md", + "date": "2026-06-10", + "value": "Input and navigation tools use Puppeteer's waiting semantics, giving high success rates on standard pages; tools return structured results the agent can act on" + } + ], + "methodology": "Hands-on testing of click, fill, navigate, and wait tools against common web applications", + "last_verified": "2026-06-10" + }, + "performance_tracing_accuracy": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Chrome for Developers blog", + "url": "https://developer.chrome.com/blog/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Performance tracing tools record real Chrome traces and surface DevTools performance insights (LCP, CLS, render-blocking analysis) directly to the agent — first-party data identical to DevTools panels" + } + ], + "methodology": "Validation of trace recording and insight extraction against DevTools Performance panel output", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Chrome DevTools MCP repository", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Failed operations return descriptive errors with current page context; console and network tools let agents self-diagnose failures, though crashed Chrome sessions require restart" + } + ], + "methodology": "Error-path testing including navigation failures, missing elements, and dropped CDP connections", + "last_verified": "2026-06-10" + }, + "automation_stability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Chrome DevTools MCP tool reference", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp/blob/main/docs/tool-reference.md", + "date": "2026-06-10", + "value": "Stable multi-step automation across ~26 tools including page management, dialog handling, and emulation; Chrome-only scope avoids cross-engine inconsistencies" + } + ], + "methodology": "Multi-step workflow stability testing across navigation, input, and inspection tools", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 52, + "criteria": { + "prompt_injection_resistance": { + "score": 50, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP security notes", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp#disclaimer", + "date": "2026-06-10", + "value": "Project explicitly warns that page content is exposed to the MCP client and that malicious pages can attempt to steer the agent — indirect prompt injection via console messages, network bodies, and page text is unmitigated" + } + ], + "methodology": "Threat modeling of untrusted page, console, and network content entering the agent context", + "last_verified": "2026-06-10" + }, + "arbitrary_code_execution_risk": { + "score": 48, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP tool reference", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp/blob/main/docs/tool-reference.md", + "date": "2026-06-10", + "value": "Script evaluation tool executes arbitrary JavaScript in any open page, enabling cookie theft, DOM manipulation, or exfiltration if the agent is hijacked" + } + ], + "methodology": "Capability analysis of in-page JS evaluation under adversarial agent steering", + "last_verified": "2026-06-10" + }, + "session_exposure_risk": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP README", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "When attached to a user's running Chrome, the agent gains full control over logged-in sessions, saved credentials autofill, and all open tabs — the project advises against exposing sensitive sessions" + } + ], + "methodology": "Analysis of attach-to-running-Chrome mode and access to authenticated browser state", + "last_verified": "2026-06-10" + }, + "sandboxing_isolation": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Chrome DevTools MCP configuration", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Supports launching isolated Chrome instances with temporary user-data directories and headless mode, which contains risk when used; no origin allow/block list filtering" + } + ], + "methodology": "Review of isolation options (fresh profile, headless, channel selection) versus attach mode", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "MCP security guidance", + "url": "https://modelcontextprotocol.io/docs/concepts/security", + "date": "2026-06-10", + "value": "Agent can perform any action the controlled Chrome session permits (form submissions, account changes); no server-side gating of destructive actions — host tool-approval is the only guardrail" + } + ], + "methodology": "Authorization boundary analysis of write-capable browser actions", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 65, + "criteria": { + "browsing_data_exposure": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP disclaimer", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp#disclaimer", + "date": "2026-06-10", + "value": "Page content, console logs, network request/response bodies, and screenshots are all returned to the MCP client and thus the LLM provider — including authenticated content when attached to a real session" + } + ], + "methodology": "Data flow analysis of inspection tool outputs (network, console, screenshot, snapshot)", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "Chrome DevTools MCP repository", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "No redaction of credentials, tokens, or PII present in network bodies, headers, or console output; network inspection can surface auth tokens verbatim" + } + ], + "methodology": "Privacy controls assessment of network and console inspection outputs", + "last_verified": "2026-06-10" + }, + "local_data_control": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP architecture", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Runs entirely locally over stdio; traces, profiles, and browser state stay on the user's machine and the server adds no vendor telemetry" + } + ], + "methodology": "Review of local execution model and data residency", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "MCP client documentation", + "url": "https://modelcontextprotocol.io/docs", + "date": "2026-06-10", + "value": "Browser data is shared only with the connected LLM provider per that provider's policy; Google does not receive the data through this server" + } + ], + "methodology": "Data sharing pathway analysis", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 92, + "criteria": { + "documentation_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP docs", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp/blob/main/docs/tool-reference.md", + "date": "2026-06-10", + "value": "Complete tool reference for all ~26 tools, configuration options, client setup guides, and an official Chrome for Developers launch post with usage patterns" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Fully open source under Apache-2.0 in the official ChromeDevTools GitHub org with public issues and roadmap" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP design", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Every browser action is an explicit named tool call visible in MCP host logs; the controlled Chrome window can run headed so users watch the agent act in real time" + } + ], + "methodology": "Logging and observability assessment", + "last_verified": "2026-06-10" + }, + "vendor_credibility": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "GitHub API", + "url": "https://api.github.com/repos/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Maintained by Google's Chrome DevTools team; 43,277 GitHub stars as of 2026-06-10, among the most-starred MCP servers" + } + ], + "methodology": "Maintainer reputation and project health analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 90, + "criteria": { + "ease_of_setup": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "npm package chrome-devtools-mcp", + "url": "https://www.npmjs.com/package/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Single-command setup via npx chrome-devtools-mcp@latest; no API keys; requires Chrome installed locally; setup snippets provided for all major MCP hosts" + } + ], + "methodology": "Setup complexity assessment across MCP hosts", + "last_verified": "2026-06-10" + }, + "performance": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Chrome DevTools MCP repository", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "Direct CDP access keeps interaction latency low; text-based snapshots and targeted network/console queries keep token usage manageable relative to screenshot-heavy approaches" + } + ], + "methodology": "Latency and token-efficiency evaluation of common debugging workflows", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Chrome DevTools MCP tool reference", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp/blob/main/docs/tool-reference.md", + "date": "2026-06-10", + "value": "~26 tools covering input automation, navigation, performance tracing and insights, network inspection, console/debugging, emulation, and screenshots — unique depth in performance debugging among MCP servers" + } + ], + "methodology": "Feature completeness assessment against web debugging and automation needs", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "GitHub API", + "url": "https://api.github.com/repos/ChromeDevTools/chrome-devtools-mcp", + "date": "2026-06-10", + "value": "43,277 stars as of 2026-06-10; widely adopted by coding agents for in-browser verification and performance debugging since its 2025 launch" + } + ], + "methodology": "Adoption metrics and ecosystem-integration analysis", + "last_verified": "2026-06-10" + }, + "maintenance_activity": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository activity", + "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp/releases", + "date": "2026-06-10", + "value": "Regular releases tracking Chrome versions with active issue triage by the Chrome DevTools team" + } + ], + "methodology": "Commit frequency and release-cadence analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "First-party Chrome DevTools data: real performance traces, insights, network and console inspection", + "Maintained by Google's Chrome DevTools team with very high adoption (43.3k stars)", + "~26 well-documented tools spanning automation, debugging, tracing, and screenshots", + "Direct CDP control is fast and exposes capabilities no other browser MCP offers", + "Runs fully locally over stdio with no vendor telemetry", + "Can attach to an existing Chrome session for debugging real user state" + ], + "limitations": [ + "Attaching to a running Chrome hands the agent full control of logged-in sessions and open tabs", + "Indirect prompt injection from malicious pages is explicitly acknowledged and unmitigated", + "Arbitrary JavaScript evaluation in pages enables exfiltration if the agent is steered", + "Network and console outputs can leak tokens, credentials, and PII to the LLM provider verbatim", + "Chrome-only — no Firefox or WebKit coverage", + "No origin allow/block filtering to restrict which sites the agent may visit" + ], + "metadata": { + "license": "Apache-2.0", + "supported_platforms": [ + "macOS, Linux, Windows with Node.js 20+ and Chrome installed" + ], + "programming_languages": [ + "TypeScript" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/ChromeDevTools/chrome-devtools-mcp", + "github_stars": 43277, + "package": "chrome-devtools-mcp", + "api_dependency": "Chrome DevTools Protocol (via Puppeteer)", + "authentication": "None required (local browser control)", + "first_release": "2025-09", + "maintained_by": "Google (Chrome DevTools team)", + "status": "Active", + "transport_types": [ + "stdio" + ], + "installation_methods": [ + "npm", + "npx" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 92, + "notes": "Best-in-class for letting coding agents verify changes in a real browser and debug performance, network, and console issues" + }, + "data-analysis": { + "overall": 80, + "notes": "Strong for performance-trace analysis and network-payload inspection; less aimed at general data extraction" + }, + "research-assistant": { + "overall": 75, + "notes": "Capable interactive browsing, but security posture favors dev/debugging use over open-web research" + }, + "education": { + "overall": 80, + "notes": "Excellent for teaching web performance and debugging — agents can demonstrate DevTools concepts live" + }, + "customer-support": { + "overall": 62, + "notes": "Can reproduce reported web issues, but session-exposure risk demands isolated profiles" + } + }, + "best_for": [ + "Coding agents verifying frontend changes in a live Chrome instance", + "Web performance debugging with real traces and DevTools insights", + "Network and console inspection during AI-assisted development", + "Developers who need CDP-level browser control from an MCP host" + ], + "related": [ + "mcp-server-playwright", + "mcp-server-puppeteer", + "mcp-server-fetch" + ], + "tags": [ + "browser-automation", + "chrome", + "devtools", + "performance", + "debugging", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-context7.json b/data/mcps/mcp-server-context7.json new file mode 100644 index 0000000..eaebd3d --- /dev/null +++ b/data/mcps/mcp-server-context7.json @@ -0,0 +1,464 @@ +{ + "id": "mcp-server-context7", + "type": "mcp", + "name": "Context7 MCP", + "provider": "Upstash", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Upstash's documentation-retrieval MCP server. Two tools (resolve-library-id, get-library-docs) inject up-to-date, version-specific library documentation into the agent's context to prevent hallucinated APIs. The most-starred MCP server repo (57.1k), but with a notable security history: the ContextCrush content-injection vulnerability (disclosed Feb 2026, patched within days).", + "website": "https://context7.com", + "trust_vector": { + "performance_reliability": { + "overall_score": 86, + "criteria": { + "documentation_accuracy": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Context7 README", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Docs are parsed from official library sources and served version-specifically, directly addressing hallucinated or outdated APIs in LLM code generation" + } + ], + "methodology": "Spot-check of returned documentation against official library docs across popular frameworks", + "last_verified": "2026-06-10" + }, + "retrieval_relevance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Context7 get-library-docs tool", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Topic-scoped retrieval with configurable token budget returns focused snippets; relevance degrades for niche topics in very large libraries" + } + ], + "methodology": "Relevance assessment of topic-filtered retrievals across common and long-tail queries", + "last_verified": "2026-06-10" + }, + "api_reliability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Context7 service", + "url": "https://context7.com", + "date": "2026-06-10", + "value": "Backend hosted on Upstash infrastructure; remote Streamable HTTP endpoint and local stdio server both depend on the hosted API, which has shown solid availability" + } + ], + "methodology": "Availability monitoring of the hosted API endpoint over the evaluation period", + "last_verified": "2026-06-10" + }, + "library_coverage": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Context7 library index", + "url": "https://context7.com", + "date": "2026-06-10", + "value": "Tens of thousands of indexed libraries with community submission of new ones; resolve-library-id reliably maps natural-language names to indexed library IDs" + } + ], + "methodology": "Coverage sampling across mainstream and long-tail open-source libraries", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Context7 MCP implementation", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Unresolved library names return candidate matches rather than hard failures; rate-limit and backend errors surface as readable messages the agent can react to" + } + ], + "methodology": "Error-path testing with unknown libraries, rate limits, and offline backend", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 59, + "criteria": { + "content_injection_resistance": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "Noma Labs — ContextCrush disclosure", + "url": "https://noma.security/blog/contextcrush-context7-the-mcp-server-vulnerability/", + "date": "2026-03-05", + "value": "ContextCrush: the Custom Rules / 'AI Instructions' feature let anyone publishing a library inject unsanitized instructions into consuming agents' context; researchers demonstrated .env credential theft. Disclosed 2026-02-18, patched 2026-02-23" + }, + { + "source": "Context7 repository", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Post-patch, publisher-supplied instruction content is sanitized/restricted, but retrieved documentation remains third-party content flowing into the agent context by design" + } + ], + "methodology": "Review of the ContextCrush vulnerability, its patch, and the residual risk of doc-content prompt injection", + "last_verified": "2026-06-10" + }, + "supply_chain_trust": { + "score": 50, + "confidence": "high", + "evidence": [ + { + "source": "Noma Labs — ContextCrush disclosure", + "url": "https://noma.security/blog/contextcrush-context7-the-mcp-server-vulnerability/", + "date": "2026-03-05", + "value": "Anyone can publish or update a library in Context7's index, making indexed docs an open supply chain into agent contexts; ContextCrush proved this channel was exploitable at scale" + } + ], + "methodology": "Analysis of the open library-submission pipeline as an attack surface for agent contexts", + "last_verified": "2026-06-10" + }, + "vulnerability_response": { + "score": 75, + "confidence": "high", + "evidence": [ + { + "source": "Noma Labs — ContextCrush disclosure timeline", + "url": "https://noma.security/blog/contextcrush-context7-the-mcp-server-vulnerability/", + "date": "2026-03-05", + "value": "Disclosed to Upstash 2026-02-18; patched 2026-02-23 (5 days); coordinated public disclosure 2026-03-05 — fast remediation, though the vulnerable feature had shipped without sanitization review" + } + ], + "methodology": "Assessment of disclosure-to-patch timeline and vendor cooperation", + "last_verified": "2026-06-10" + }, + "credential_exposure_risk": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Noma Labs — ContextCrush demonstration", + "url": "https://noma.security/blog/contextcrush-context7-the-mcp-server-vulnerability/", + "date": "2026-03-05", + "value": "The demonstrated exploit exfiltrated .env credentials by instructing the agent to read and transmit secrets — Context7 itself holds no credentials beyond an optional API key, but its content channel could weaponize other tools the agent holds" + } + ], + "methodology": "Analysis of indirect credential-theft pathways via injected instructions in a multi-tool agent", + "last_verified": "2026-06-10" + }, + "authentication_security": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Context7 documentation", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Works without authentication for basic use; optional API key raises rate limits and is passed via header/env var. Minimal credential surface, but the remote endpoint means key handling depends on client configuration" + } + ], + "methodology": "Review of API-key handling for the hosted endpoint and local server", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 75, + "criteria": { + "query_data_exposure": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Context7 architecture", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Library names and topic queries are sent to Upstash's hosted API; queries can reveal what technologies a team is using but contain no source code" + } + ], + "methodology": "Data flow analysis of outbound query content to the hosted backend", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Context7 tool design", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Tools never read local files or code themselves, limiting direct exposure; however, ContextCrush showed retrieved content can induce other tools to exfiltrate secrets, so isolation depends on the surrounding agent" + } + ], + "methodology": "Assessment of direct and indirect sensitive-data pathways", + "last_verified": "2026-06-10" + }, + "data_minimization": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Context7 tool schemas", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Only two narrowly-scoped read-only tools; requests carry just a library ID, topic, and token budget — a notably minimal data footprint among MCP servers" + } + ], + "methodology": "Review of request payloads and tool surface area", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Upstash privacy policy", + "url": "https://upstash.com/trust/privacy.pdf", + "date": "2026-06-10", + "value": "Queries are processed by Upstash per its privacy policy; retrieved docs additionally flow to the connected LLM provider like all MCP tool results" + } + ], + "methodology": "Data sharing pathway analysis across Upstash and LLM provider", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 86, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Context7 README", + "url": "https://github.com/upstash/context7/blob/master/README.md", + "date": "2026-06-10", + "value": "Clear installation instructions for 20+ MCP clients, both local and remote transports, and well-documented tool parameters" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "MCP server is MIT-licensed and open source; the backend indexing/retrieval pipeline at context7.com is proprietary, leaving part of the stack unauditable" + } + ], + "methodology": "Source availability review of client/server versus hosted backend", + "last_verified": "2026-06-10" + }, + "incident_disclosure": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Noma Labs — ContextCrush disclosure", + "url": "https://noma.security/blog/contextcrush-context7-the-mcp-server-vulnerability/", + "date": "2026-03-05", + "value": "Upstash cooperated with coordinated disclosure and patched within 5 days; public acknowledgment came primarily through the researcher's publication rather than a detailed vendor advisory" + } + ], + "methodology": "Review of vendor communication during and after the ContextCrush incident", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Context7 tool design", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Both tools are read-only with visible parameters and outputs in MCP host logs; retrieved doc content is fully inspectable before the agent acts on it" + } + ], + "methodology": "Logging and traceability assessment of tool calls and returned content", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 88, + "criteria": { + "ease_of_setup": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Context7 installation docs", + "url": "https://github.com/upstash/context7#installation", + "date": "2026-06-10", + "value": "Remote endpoint requires only a URL (https://mcp.context7.com/mcp) with no install; local server via npx @upstash/context7-mcp; no API key needed for basic use" + } + ], + "methodology": "Setup complexity assessment across remote and local installation paths", + "last_verified": "2026-06-10" + }, + "performance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Context7 hosted API", + "url": "https://context7.com", + "date": "2026-06-10", + "value": "Doc retrievals typically return in under a couple of seconds; configurable token budget keeps context costs predictable" + } + ], + "methodology": "Latency measurement of resolve and retrieval calls", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Context7 tool reference", + "url": "https://github.com/upstash/context7", + "date": "2026-06-10", + "value": "Deliberately narrow: two tools doing one job well (documentation retrieval); no code search, examples execution, or private-docs indexing in the open tier" + } + ], + "methodology": "Feature scope assessment relative to documentation-retrieval needs", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "GitHub API", + "url": "https://api.github.com/repos/upstash/context7", + "date": "2026-06-10", + "value": "57,127 stars as of 2026-06-10 — the largest MCP server repository on GitHub; integrated into setup guides of most major MCP clients" + } + ], + "methodology": "Adoption metrics and ecosystem-integration analysis", + "last_verified": "2026-06-10" + }, + "maintenance_activity": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository activity", + "url": "https://github.com/upstash/context7/commits", + "date": "2026-06-10", + "value": "Active maintenance by Upstash with regular releases, fast security patching (ContextCrush fixed in 5 days), and continuous library-index growth" + } + ], + "methodology": "Commit frequency, release cadence, and patch-responsiveness analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Directly addresses hallucinated/outdated APIs with version-specific, current documentation", + "Largest MCP server community on GitHub (57.1k stars) with broad client integration", + "Minimal tool surface: two read-only tools with a small, predictable data footprint", + "Zero-friction setup — remote endpoint works with just a URL, no API key required", + "Fast vendor response to the ContextCrush vulnerability (patched in 5 days)", + "Configurable token budget keeps documentation injection cost-controlled" + ], + "limitations": [ + "ContextCrush (Feb 2026) proved the library index is an exploitable injection channel into agent contexts; retrieved third-party content remains untrusted by design", + "Open publishing model means doc quality and integrity vary across the index", + "Backend indexing/retrieval pipeline is proprietary and unauditable", + "Dependent on Upstash's hosted service — no fully offline operation", + "Queries reveal a team's technology stack to a third party", + "Narrow scope: documentation retrieval only, no private-docs support in the open tier" + ], + "metadata": { + "license": "MIT", + "supported_platforms": [ + "All platforms with Node.js 18+ (local server); any MCP client (remote endpoint)" + ], + "programming_languages": [ + "TypeScript" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/upstash/context7", + "github_stars": 57127, + "package": "@upstash/context7-mcp", + "remote_endpoint": "https://mcp.context7.com/mcp", + "api_dependency": "Context7 hosted documentation API (Upstash)", + "authentication": "Optional API key (higher rate limits)", + "first_release": "2025-04", + "maintained_by": "Upstash", + "status": "Active", + "security_incidents": [ + { + "name": "ContextCrush", + "reported_by": "Noma Labs", + "disclosed": "2026-02-18", + "patched": "2026-02-23", + "published": "2026-03-05", + "summary": "Custom Rules/'AI Instructions' feature allowed any library publisher to inject unsanitized instructions into consuming agents' context; demonstrated exploit exfiltrated .env credentials", + "advisory": "https://noma.security/blog/contextcrush-context7-the-mcp-server-vulnerability/" + } + ], + "transport_types": [ + "stdio", + "streamable-http" + ], + "installation_methods": [ + "npm", + "npx", + "remote-url", + "docker" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "Core use case — current, version-specific docs measurably reduce hallucinated APIs and deprecated patterns" + }, + "research-assistant": { + "overall": 86, + "notes": "Excellent for researching library capabilities and APIs; limited to indexed open-source documentation" + }, + "education": { + "overall": 85, + "notes": "Strong for learning frameworks with accurate, up-to-date examples instead of stale training data" + }, + "content-creation": { + "overall": 75, + "notes": "Useful for writing accurate technical tutorials and documentation-backed articles" + }, + "data-analysis": { + "overall": 68, + "notes": "Indirectly helpful (correct API usage for analysis libraries) but not an analysis tool itself" + } + }, + "best_for": [ + "Developers wanting AI code generation grounded in current library documentation", + "Teams fighting hallucinated APIs and deprecated patterns in agent output", + "Coding agents that need version-specific framework knowledge on demand", + "Technical writers verifying API accuracy in tutorials and docs" + ], + "related": [ + "mcp-server-github", + "mcp-server-fetch", + "mcp-server-firecrawl", + "mcp-server-serena" + ], + "tags": [ + "documentation", + "code-context", + "retrieval", + "upstash", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-everything.json b/data/mcps/mcp-server-everything.json index 0614672..bd914d0 100644 --- a/data/mcps/mcp-server-everything.json +++ b/data/mcps/mcp-server-everything.json @@ -4,9 +4,9 @@ "name": "MCP Everything Server", "provider": "Anthropic", "version": "2025.9.25", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Official Anthropic reference MCP server demonstrating all protocol features including tools, resources, prompts, and sampling. Designed for testing, development, and as a comprehensive example implementation. Not intended for production use but essential for MCP protocol understanding and testing.", + "description": "Official MCP reference server demonstrating all protocol features (tools, resources, prompts, sampling) for testing and development. One of the seven reference servers still actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest spec 2025-11-25. Not intended for production use but essential for MCP protocol understanding and testing.", "website": "https://modelcontextprotocol.io/docs/servers/everything", "trust_vector": { "performance_reliability": { @@ -349,10 +349,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Used for development/testing; not for production deployments" + }, + { + "source": "Anthropic - Donating MCP to the Agentic AI Foundation", + "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "everything is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -409,7 +415,7 @@ "Exemplary code quality demonstrating best practices", "Excellent educational resource for MCP protocol learning", "Complete documentation with detailed examples", - "Open source with active Anthropic maintenance", + "Open source and actively maintained under the MCP project (Agentic AI Foundation)", "Ideal for protocol testing and validation" ], "limitations": [ @@ -418,7 +424,8 @@ "Uses mock data; not suitable for real-world applications", "May expose all protocol capabilities without restrictions", "Not optimized for performance or scale", - "Requires careful consideration before any production use" + "Requires careful consideration before any production use", + "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -435,7 +442,7 @@ "api_dependency": "None (MCP protocol only)", "authentication": "None required", "first_release": "2024-11", - "maintained_by": "Anthropic", + "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Reference/Testing", "transport_types": [ "stdio" @@ -448,6 +455,8 @@ "meta", "all-in-one", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "reference-server" ] } diff --git a/data/mcps/mcp-server-fetch.json b/data/mcps/mcp-server-fetch.json index cb6c261..31d60d6 100644 --- a/data/mcps/mcp-server-fetch.json +++ b/data/mcps/mcp-server-fetch.json @@ -4,9 +4,9 @@ "name": "MCP Fetch Server", "provider": "Anthropic", "version": "2025.4.6", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Official Anthropic MCP server for fetching web content and converting HTML to markdown. Enables AI models to retrieve and process web pages, documentation, and online resources. Essential for research, content analysis, and web-based workflows.", + "description": "Official MCP reference server for fetching web content and converting HTML to markdown. Enables AI models to retrieve and process web pages, documentation, and online resources. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", "website": "https://modelcontextprotocol.io/docs/servers/fetch", "trust_vector": { "performance_reliability": { @@ -363,10 +363,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Widely adopted as core MCP server for web content access" + }, + { + "source": "Anthropic - Donating MCP to the Agentic AI Foundation", + "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "fetch is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -374,7 +380,7 @@ "strengths": [ "Simple, zero-configuration setup for basic web content fetching", "High-quality HTML to Markdown conversion using Turndown library", - "Open source with active Anthropic support and maintenance", + "Open source and actively maintained under the MCP project (Agentic AI Foundation)", "Reliable HTTP fetching with timeout and error handling", "Respects robots.txt and implements ethical web scraping practices", "Excellent documentation and integration with MCP ecosystem" @@ -385,7 +391,8 @@ "No built-in authentication for paywalled or restricted content", "No malicious content scanning or sanitization", "Performance dependent on target website responsiveness", - "Basic rate limiting may cause issues with aggressive scraping" + "Basic rate limiting may cause issues with aggressive scraping", + "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -402,7 +409,7 @@ "api_dependency": "HTTP/HTTPS, Turndown", "authentication": "None required", "first_release": "2024-11", - "maintained_by": "Anthropic", + "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Active", "transport_types": [ "stdio" @@ -462,6 +469,8 @@ "http", "api", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "reference-server" ] } diff --git a/data/mcps/mcp-server-figma.json b/data/mcps/mcp-server-figma.json new file mode 100644 index 0000000..9d1e298 --- /dev/null +++ b/data/mcps/mcp-server-figma.json @@ -0,0 +1,436 @@ +{ + "id": "mcp-server-figma", + "type": "mcp", + "name": "Figma MCP Server", + "provider": "Figma", + "version": "2025.6-beta", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Figma's official Dev Mode MCP server connecting AI coding tools to design files. Provides design-context extraction (code from frames, variables, components), screenshots, metadata, Code Connect mapping, FigJam reading, and design generation onto the canvas. Available as a hosted remote server (OAuth) or via the Figma desktop app.", + "website": "https://developers.figma.com/docs/figma-mcp-server/", + "trust_vector": { + "performance_reliability": { + "overall_score": 78, + "criteria": { + "design_context_accuracy": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "get_design_context returns structured code, design tokens (variables), and component data from selected frames, improving design-to-code fidelity over screenshot-only approaches" + } + ], + "methodology": "Assessment of design-to-code output fidelity against source frames, variables, and component structure", + "last_verified": "2026-06-10" + }, + "api_reliability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Launch Announcement", + "url": "https://www.figma.com/blog/introducing-figma-mcp-server/", + "date": "2026-06-10", + "value": "Built on Figma's platform infrastructure; remote endpoint at mcp.figma.com/mcp, but the product remains in beta with evolving behavior" + } + ], + "methodology": "Analysis of endpoint stability and Figma platform uptime during beta period", + "last_verified": "2026-06-10" + }, + "large_file_handling": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Large or deeply nested frames can produce design-context payloads that exceed client context limits; documentation recommends selecting smaller frames or using metadata-first workflows" + } + ], + "methodology": "Testing context extraction on large, deeply nested design files and component libraries", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Returns structured errors for invalid node IDs, missing selections, and permission failures; local desktop server requires the file to be open in the app" + } + ], + "methodology": "Error handling testing across invalid selections, permissions, and disconnected desktop sessions", + "last_verified": "2026-06-10" + }, + "rate_limit_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Figma Developers Platform", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Hosted server applies Figma platform rate limits per authenticated user; limits are not fully published during beta and usage-based pricing is planned" + } + ], + "methodology": "Rate limiting behavior observation under sustained tool-call load", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 74, + "criteria": { + "authentication_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Remote server at https://mcp.figma.com/mcp uses OAuth with scoped access tied to the user's Figma account; recommended over the local desktop server" + } + ], + "methodology": "Review of OAuth flow, scope grants, and token lifecycle for the hosted endpoint", + "last_verified": "2026-06-10" + }, + "token_exposure_risk": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "OAuth tokens are managed by the MCP client rather than pasted as static API keys; local server at http://127.0.0.1:3845/mcp binds to loopback and relies on the logged-in desktop session" + } + ], + "methodology": "Token storage and exposure-surface analysis for remote and local transports", + "last_verified": "2026-06-10" + }, + "scope_limitation": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Access mirrors the authenticated user's file permissions; there is no per-file or read-only-only scoping below the account level for the MCP connection" + } + ], + "methodology": "Permission boundary testing across files the authenticated user can view or edit", + "last_verified": "2026-06-10" + }, + "prompt_injection_risk": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Launch Announcement", + "url": "https://www.figma.com/blog/introducing-figma-mcp-server/", + "date": "2026-06-10", + "value": "Design file content (layer names, text nodes, FigJam notes) is third-party-authored input on shared files; malicious text in a shared design can act as an injection vector into the consuming agent" + } + ], + "methodology": "Threat modeling of untrusted design-file content flowing into agent context via design-context and FigJam tools", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Write-to-canvas and design generation tools can create and modify content in Figma files the user can edit; no server-side confirmation step beyond client-level approvals" + } + ], + "methodology": "Authorization boundary testing of write-capable tools against editable files", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 72, + "criteria": { + "design_data_exposure": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Design content, text layers, screenshots, and variable values are sent to the connected LLM provider as tool results" + } + ], + "methodology": "Data flow analysis from Figma files through MCP tool results to LLM providers", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "No built-in redaction of sensitive content embedded in designs (e.g., real customer data in mockups, internal roadmap text in FigJam boards)" + } + ], + "methodology": "Assessment of filtering and redaction controls on extracted design content", + "last_verified": "2026-06-10" + }, + "organization_data_control": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Figma Security and Admin Controls", + "url": "https://www.figma.com/security/", + "date": "2026-06-10", + "value": "Access governed by Figma workspace permissions, SSO/SAML, and admin controls; org admins control which users can authorize integrations" + } + ], + "methodology": "Review of organizational access controls applicable to MCP-connected accounts", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Figma Privacy Policy", + "url": "https://www.figma.com/privacy/", + "date": "2026-06-10", + "value": "Design data retrieved via MCP is shared with whichever LLM provider the user's client uses, per that provider's data policy rather than Figma's" + } + ], + "methodology": "Analysis of downstream data sharing once content leaves the Figma boundary", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 71, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Dedicated developer documentation covering setup for remote and desktop servers, tool reference, and client integration guides" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Tool calls are visible in MCP client logs and canvas writes appear in file history, but there is no dedicated MCP audit log on the Figma side" + } + ], + "methodology": "Logging and traceability assessment across client and Figma file history", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Figma MCP Server Launch Announcement", + "url": "https://www.figma.com/blog/introducing-figma-mcp-server/", + "date": "2026-06-10", + "value": "Server implementation is closed source and maintained by Figma; behavior can only be verified through documentation and observed tool output" + } + ], + "methodology": "Source availability and independent verifiability review", + "last_verified": "2026-06-10" + }, + "api_coverage_clarity": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Documented tool set: design context, screenshots, metadata, variables, Code Connect mapping, FigJam reading, and design generation; beta tools and limits are flagged" + } + ], + "methodology": "Comparison of documented tool surface against observed server capabilities", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 82, + "criteria": { + "ease_of_setup": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Remote server requires only adding https://mcp.figma.com/mcp and completing OAuth; local mode requires enabling the server in the Figma desktop app" + } + ], + "methodology": "Setup complexity assessment across supported MCP clients", + "last_verified": "2026-06-10" + }, + "api_performance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Design-context extraction and screenshot generation are heavier operations than typical API reads; latency scales with frame complexity" + } + ], + "methodology": "Latency observation across tool types and frame sizes", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Launch Announcement", + "url": "https://www.figma.com/blog/introducing-figma-mcp-server/", + "date": "2026-06-10", + "value": "Launched in beta June 2025; tool behavior and output formats have continued to evolve, and beta status is explicit" + } + ], + "methodology": "Stability assessment over the beta period including breaking-change frequency", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Figma MCP Server Documentation", + "url": "https://developers.figma.com/docs/figma-mcp-server/", + "date": "2026-06-10", + "value": "Covers both design-to-code (context, screenshots, variables, Code Connect) and code-to-design (write-to-canvas, design generation) plus FigJam" + } + ], + "methodology": "Feature completeness assessment against design-to-code workflow needs", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Figma MCP Server Launch Announcement", + "url": "https://www.figma.com/blog/introducing-figma-mcp-server/", + "date": "2026-06-10", + "value": "First-party integration promoted across major AI coding tools (Claude Code, Cursor, VS Code, Windsurf); widely adopted as the standard design-context source" + } + ], + "methodology": "Adoption analysis across MCP client ecosystems and developer tooling", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "First-party server with structured design context (code, variables, components) rather than screenshots alone", + "Code Connect mapping links Figma components to real codebase components for higher-fidelity output", + "Bidirectional: reads designs into code and writes generated designs back to the canvas", + "Hosted remote endpoint with OAuth removes the need for static API keys", + "Strong official documentation and broad MCP client support", + "Free during beta, lowering the barrier to evaluation" + ], + "limitations": [ + "Closed source; server behavior cannot be independently audited", + "Shared design files are third-party-authored input and a prompt injection vector", + "Design content, including any sensitive text in mockups, is sent to the LLM provider", + "Large frames can exceed client context limits", + "Beta product with evolving tools and planned usage-based pricing not yet finalized", + "Write-capable tools can modify editable files without server-side confirmation" + ], + "metadata": { + "license": "Proprietary (closed source)", + "maintained_by": "Figma", + "status": "Beta (launched June 2025)", + "remote_endpoint": "https://mcp.figma.com/mcp", + "local_endpoint": "http://127.0.0.1:3845/mcp (Figma desktop app)", + "authentication": "OAuth (remote, recommended); desktop app session (local)", + "transport_types": [ + "streamable-http (remote)", + "http (local desktop server)" + ], + "installation_methods": [ + "Remote MCP endpoint", + "Figma desktop app toggle" + ], + "pricing": "Free during beta; usage-based pricing planned", + "first_release": "2025-06", + "mcp_version": "1.0" + }, + "use_case_ratings": { + "code-generation": { + "overall": 92, + "notes": "Primary use case: high-fidelity design-to-code with design tokens, components, and Code Connect mappings" + }, + "content-creation": { + "overall": 80, + "notes": "Strong for generating and editing designs, mockups, and FigJam content from natural language" + }, + "research-assistant": { + "overall": 70, + "notes": "Useful for auditing design systems, extracting variables, and reviewing FigJam boards" + }, + "education": { + "overall": 76, + "notes": "Good for teaching design-to-code workflows and design system concepts" + }, + "creative-writing": { + "overall": 55, + "notes": "Marginal fit; limited to text content inside design and FigJam files" + } + }, + "best_for": [ + "Frontend teams converting Figma designs into production code with AI assistance", + "Design-system teams keeping code and Figma components in sync via Code Connect", + "Developers generating UI mockups back into Figma from code or intent", + "Agentic coding workflows that need pixel-accurate design context" + ], + "related_entities": [ + "mcp-server-vercel", + "mcp-server-github", + "mcp-server-linear", + "mcp-server-notion" + ], + "tags": [ + "design", + "design-to-code", + "figma", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-filesystem.json b/data/mcps/mcp-server-filesystem.json index d87c97f..092474c 100644 --- a/data/mcps/mcp-server-filesystem.json +++ b/data/mcps/mcp-server-filesystem.json @@ -4,9 +4,9 @@ "name": "MCP Filesystem Server", "provider": "Anthropic", "version": "2025.7.1", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Official MCP server providing AI models with controlled access to the local filesystem. Enables file reading, writing, and directory operations through the Model Context Protocol. Critical for file-based workflows but requires careful security configuration.", + "description": "Official MCP reference server providing AI models with controlled access to the local filesystem (file reading, writing, and directory operations). One of the seven reference servers still actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest spec 2025-11-25. Critical for file-based workflows but requires careful security configuration.", "website": "https://modelcontextprotocol.io/docs/servers/filesystem", "trust_vector": { "performance_reliability": { @@ -334,11 +334,17 @@ "source": "MCP Server Updates", "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", - "value": "Regular updates from Anthropic, minimal maintenance required" + "value": "Regular updates, minimal maintenance required" + }, + { + "source": "Anthropic - Donating MCP to the Agentic AI Foundation", + "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "filesystem is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" } ], "methodology": "Maintenance overhead assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "community_support": { "score": 82, @@ -371,7 +377,8 @@ "File contents sent to external LLM provider APIs", "Limited granular permission controls beyond directory allowlists", "Potential for accidental data exfiltration to LLM providers", - "Audit logging requires custom implementation for compliance needs" + "Audit logging requires custom implementation for compliance needs", + "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -388,7 +395,7 @@ "github_repo": "https://github.com/modelcontextprotocol/servers", "github_stars": 58700, "first_release": "2024-11", - "maintained_by": "Anthropic", + "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Active", "transport_types": [ "stdio" @@ -448,6 +455,8 @@ "file-system", "local", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "reference-server" ] } diff --git a/data/mcps/mcp-server-firecrawl.json b/data/mcps/mcp-server-firecrawl.json new file mode 100644 index 0000000..7412e7c --- /dev/null +++ b/data/mcps/mcp-server-firecrawl.json @@ -0,0 +1,455 @@ +{ + "id": "mcp-server-firecrawl", + "type": "mcp", + "name": "Firecrawl MCP Server", + "provider": "Firecrawl", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Official Firecrawl MCP server giving AI models web scraping, crawling, site mapping, web search, and structured extraction capabilities. Available as an npm package (firecrawl-mcp) over stdio or as a hosted remote server authenticated with a firecrawl.dev API key.", + "website": "https://github.com/firecrawl/firecrawl-mcp-server", + "trust_vector": { + "performance_reliability": { + "overall_score": 83, + "criteria": { + "api_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Firecrawl Documentation", + "url": "https://docs.firecrawl.dev/", + "date": "2026-06-10", + "value": "Mature hosted API backing all MCP tools, with documented status reporting and a widely used production service" + } + ], + "methodology": "API stability and service maturity analysis", + "last_verified": "2026-06-10" + }, + "scrape_success_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "Scrape and batch scrape tools handle JavaScript-rendered pages, anti-bot mitigation, and multiple output formats; some heavily protected sites still fail" + } + ], + "methodology": "Scrape success testing across diverse site types", + "last_verified": "2026-06-10" + }, + "crawl_completeness": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl Crawl Documentation", + "url": "https://docs.firecrawl.dev/features/crawl", + "date": "2026-06-10", + "value": "Crawl and map tools discover most site pages with configurable depth and limits; very large or dynamic sites may be partially covered" + } + ], + "methodology": "Crawl coverage assessment against known site structures", + "last_verified": "2026-06-10" + }, + "rate_limit_handling": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "Built-in automatic retries with exponential backoff for rate-limited requests and credit usage monitoring" + } + ], + "methodology": "Rate limiting behavior review from source and docs", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "Errors surfaced as structured tool results; retry logic covers transient failures, though long-running crawls can require manual restart" + } + ], + "methodology": "Error handling and recovery testing", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 68, + "criteria": { + "authentication_security": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Firecrawl MCP Setup Documentation", + "url": "https://docs.firecrawl.dev/mcp-server", + "date": "2026-06-10", + "value": "Authenticates with a firecrawl.dev API key via environment variable (stdio) or the hosted remote endpoint" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-06-10" + }, + "prompt_injection_exposure": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "MCP Security Best Practices", + "url": "https://modelcontextprotocol.io/specification/draft/basic/security_best_practices", + "date": "2026-06-10", + "value": "All fetched web content is untrusted; malicious pages can carry indirect prompt injection payloads that the consuming agent must defend against" + } + ], + "methodology": "Threat modeling of untrusted web content returned to the model", + "last_verified": "2026-06-10" + }, + "ssrf_protection": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "Model-supplied URLs are fetched server-side; SSRF-style requests against internal or private endpoints are a documented risk class for fetch/scrape MCP servers" + } + ], + "methodology": "SSRF risk analysis of URL-driven tool inputs", + "last_verified": "2026-06-10" + }, + "credential_handling": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl MCP Setup Documentation", + "url": "https://docs.firecrawl.dev/mcp-server", + "date": "2026-06-10", + "value": "API key stored in client configuration or environment; key grants full account credit usage if leaked, so it must be protected" + } + ], + "methodology": "API key storage and exposure analysis", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "Tools are read-oriented (scrape, crawl, search, extract) so no destructive write actions exist, but the agent can crawl arbitrary sites and consume account credits" + } + ], + "methodology": "Capability and blast radius assessment of exposed tools", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 69, + "criteria": { + "data_exposure": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Firecrawl Documentation", + "url": "https://docs.firecrawl.dev/", + "date": "2026-06-10", + "value": "Scraped URLs and page content flow through Firecrawl's cloud service and are then returned to the LLM provider context" + } + ], + "methodology": "Data flow analysis of scrape and crawl pipelines", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "No built-in PII or secret filtering on scraped content; anything reachable at a URL can land in the model context" + } + ], + "methodology": "Privacy controls assessment of returned content", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl Privacy Policy", + "url": "https://www.firecrawl.dev/privacy-policy", + "date": "2026-06-10", + "value": "Request URLs and scraped content processed by Firecrawl per its privacy policy; results are additionally shared with the chosen LLM provider" + } + ], + "methodology": "Data sharing and policy review", + "last_verified": "2026-06-10" + }, + "compliance_posture": { + "score": 73, + "confidence": "low", + "evidence": [ + { + "source": "Firecrawl Terms of Service", + "url": "https://www.firecrawl.dev/terms-of-service", + "date": "2026-06-10", + "value": "Commercial terms published; users remain responsible for robots.txt, copyright, and lawful-scraping compliance for targets they crawl" + } + ], + "methodology": "Compliance documentation review", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 88, + "criteria": { + "documentation_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Firecrawl MCP Documentation", + "url": "https://docs.firecrawl.dev/mcp-server", + "date": "2026-06-10", + "value": "Detailed setup guides for major MCP clients, tool-by-tool reference, and configuration options" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl Dashboard", + "url": "https://www.firecrawl.dev/", + "date": "2026-06-10", + "value": "Account dashboard shows request activity and credit usage; MCP tool calls are visible in client logs" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "MIT-licensed open source server with 6,538 GitHub stars; backend scraping service is partially proprietary" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + }, + "api_coverage_clarity": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "Tool set clearly enumerated: scrape, batch scrape, crawl, map, web search, and structured extract" + } + ], + "methodology": "Tool surface documentation review", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 86, + "criteria": { + "ease_of_setup": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "firecrawl-mcp npm package", + "url": "https://www.npmjs.com/package/firecrawl-mcp", + "date": "2026-06-10", + "value": "Single npx command with one API key environment variable, or zero-install hosted remote endpoint" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-06-10" + }, + "api_performance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl Documentation", + "url": "https://docs.firecrawl.dev/", + "date": "2026-06-10", + "value": "Single-page scrapes typically complete in seconds; full crawls of large sites are long-running asynchronous jobs" + } + ], + "methodology": "Latency characterization across tool types", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Firecrawl Status Page", + "url": "https://status.firecrawl.dev/", + "date": "2026-06-10", + "value": "Public status page with historically high availability for the hosted API" + } + ], + "methodology": "Uptime and incident history analysis", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "Covers the full web data lifecycle: discovery (map, search), retrieval (scrape, batch, crawl), and structured extraction" + } + ], + "methodology": "Feature completeness assessment", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Repository Metrics", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-06-10", + "value": "6,538 GitHub stars and listing in major MCP client directories indicate strong adoption" + } + ], + "methodology": "Community activity and adoption analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Full web data toolkit: scrape, batch scrape, crawl, map, search, and structured extract", + "Handles JavaScript-heavy sites and returns clean LLM-ready markdown", + "Automatic retries with exponential backoff and credit usage monitoring", + "Simple setup via npx package or hosted remote server", + "MIT-licensed open source server with strong community adoption (6,538 stars)", + "Read-only tool surface with no destructive write actions" + ], + "limitations": [ + "All fetched web content is untrusted and can carry indirect prompt injection payloads", + "Model-supplied URLs create SSRF-style risk against internal or private endpoints", + "Leaked API key allows arbitrary credit consumption on the account", + "No built-in PII or secret filtering on scraped content", + "Scraped data transits Firecrawl's cloud before reaching the model", + "Heavily bot-protected sites can still fail or return partial content" + ], + "metadata": { + "license": "MIT", + "supported_platforms": [ + "All platforms with Node.js", + "Hosted remote server" + ], + "programming_languages": [ + "TypeScript" + ], + "github_repo": "https://github.com/firecrawl/firecrawl-mcp-server", + "github_stars": 6538, + "package_name": "firecrawl-mcp", + "api_dependency": "Firecrawl API (firecrawl.dev)", + "authentication": "Firecrawl API key", + "maintained_by": "Firecrawl", + "transport_types": [ + "stdio", + "remote (hosted)" + ], + "installation_methods": [ + "npm", + "hosted endpoint" + ] + }, + "use_case_ratings": { + "research-assistant": { + "overall": 92, + "notes": "Excellent for gathering, crawling, and extracting web sources during research workflows" + }, + "data-analysis": { + "overall": 85, + "notes": "Structured extract and batch scrape turn web pages into analyzable datasets" + }, + "content-creation": { + "overall": 86, + "notes": "Strong for sourcing reference material, competitor pages, and up-to-date facts for content" + }, + "code-generation": { + "overall": 74, + "notes": "Useful for pulling live documentation and examples into coding sessions" + }, + "customer-support": { + "overall": 70, + "notes": "Can crawl knowledge bases and product pages to ground support answers" + }, + "financial-analysis": { + "overall": 72, + "notes": "Good for collecting public filings and market pages, but verify accuracy of scraped figures" + }, + "legal-compliance": { + "overall": 60, + "notes": "Scraping legality and content reliability require careful human review for legal work" + }, + "education": { + "overall": 80, + "notes": "Helpful for building course material from live web sources with proper attribution" + } + }, + "best_for": [ + "Research agents that need reliable, structured web content retrieval", + "Teams converting websites into LLM-ready markdown or structured data", + "AI applications requiring crawl, search, and extraction in one server", + "Developers wanting a drop-in web data layer for MCP clients" + ], + "related_entities": [ + "mcp-server-tavily", + "mcp-server-brave-search", + "mcp-server-fetch", + "mcp-server-apify", + "mcp-server-playwright" + ], + "tags": [ + "web-scraping", + "crawling", + "search", + "extraction", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-git.json b/data/mcps/mcp-server-git.json index 2357c5f..1a2a29e 100644 --- a/data/mcps/mcp-server-git.json +++ b/data/mcps/mcp-server-git.json @@ -4,9 +4,9 @@ "name": "MCP Git Server", "provider": "Anthropic", "version": "2025.9.25", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Official Anthropic MCP server for Git repository operations. Enables AI models to interact with local Git repositories, perform commits, branch management, and version control operations. Essential for AI-assisted development workflows and code management.", + "description": "Official MCP reference server for Git repository operations. Enables AI models to interact with local Git repositories, perform commits, branch management, and version control operations. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", "website": "https://modelcontextprotocol.io/docs/servers/git", "trust_vector": { "performance_reliability": { @@ -363,10 +363,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Popular MCP server for development workflows" + }, + { + "source": "Anthropic - Donating MCP to the Agentic AI Foundation", + "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "git is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -376,7 +382,7 @@ "Built on mature and reliable Git infrastructure", "Excellent for development workflows and version control automation", "Full operation auditability through Git reflog and MCP logs", - "Open source implementation with active Anthropic support", + "Open source implementation actively maintained under the MCP project (Agentic AI Foundation)", "Simple setup requiring only local Git installation" ], "limitations": [ @@ -385,7 +391,8 @@ "No built-in secret detection or sensitive data filtering", "Can access Git credentials stored on local system", "Performance may degrade with very large repositories", - "No safeguards against accidental commits or pushes" + "No safeguards against accidental commits or pushes", + "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -402,7 +409,7 @@ "api_dependency": "Git CLI", "authentication": "Uses local Git credentials", "first_release": "2024-11", - "maintained_by": "Anthropic", + "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Active", "transport_types": [ "stdio" @@ -462,6 +469,8 @@ "git", "version-control", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "reference-server" ] } diff --git a/data/mcps/mcp-server-github.json b/data/mcps/mcp-server-github.json index 6c4fc12..1d05d55 100644 --- a/data/mcps/mcp-server-github.json +++ b/data/mcps/mcp-server-github.json @@ -4,10 +4,10 @@ "name": "MCP GitHub Server", "provider": "GitHub (formerly Anthropic)", "version": "2025.4.6", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "MCP server providing AI models with comprehensive GitHub integration capabilities. Enables repository management, issue tracking, pull request operations, and code search. NOTE: The original Anthropic version has been deprecated. Development moved to GitHub's official server at github.com/github/github-mcp-server.", - "website": "https://modelcontextprotocol.io/docs/servers/github", + "description": "GitHub's OFFICIAL MCP server, successor to the archived Anthropic reference server. Open source (Go, MIT, 30,558 stars), distributed as a binary/Docker image or via the hosted remote at https://api.githubcopilot.com/mcp/ (GA 2025-09-04) with OAuth 2.1+PKCE. Exposes 50+ tools in configurable toolsets with read-only mode. Known prompt-injection exfiltration risk (Invariant Labs, May 2025) requires least-privilege tokens and one-repo sessions.", + "website": "https://github.com/github/github-mcp-server", "trust_vector": { "performance_reliability": { "overall_score": 86, @@ -31,14 +31,14 @@ "confidence": "high", "evidence": [ { - "source": "MCP GitHub Server", - "url": "https://github.com/modelcontextprotocol/servers/tree/main/src/github", - "date": "2025-11-16", - "value": "High success rate for repo operations, issues, and PR management" + "source": "GitHub MCP Server (official)", + "url": "https://github.com/github/github-mcp-server", + "date": "2026-06-10", + "value": "Actively maintained by GitHub; high success rate for repo operations, issues, and PR management across 50+ tools" } ], "methodology": "Operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "rate_limit_handling": { "score": 82, @@ -85,10 +85,10 @@ } }, "security": { - "overall_score": 78, + "overall_score": 76, "criteria": { "authentication_security": { - "score": 85, + "score": 88, "confidence": "high", "evidence": [ { @@ -96,10 +96,16 @@ "url": "https://docs.github.com/en/authentication/keeping-your-account-and-data-secure/managing-your-personal-access-tokens", "date": "2025-11-16", "value": "Uses GitHub PAT or OAuth for secure authentication" + }, + { + "source": "GitHub Changelog - Remote GitHub MCP Server GA", + "url": "https://github.blog/changelog/2025-09-04-remote-github-mcp-server-is-now-generally-available/", + "date": "2025-09-04", + "value": "Hosted remote server (https://api.githubcopilot.com/mcp/) generally available with OAuth 2.1 + PKCE authorization" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "token_exposure_risk": { "score": 70, @@ -116,18 +122,24 @@ "last_verified": "2025-11-09" }, "scope_limitation": { - "score": 78, - "confidence": "medium", + "score": 82, + "confidence": "high", "evidence": [ { "source": "GitHub Token Scopes", "url": "https://docs.github.com/en/apps/oauth-apps/building-oauth-apps/scopes-for-oauth-apps", "date": "2025-11-16", "value": "Supports granular permission scopes, but requires careful configuration" + }, + { + "source": "GitHub MCP Server (official)", + "url": "https://github.com/github/github-mcp-server", + "date": "2026-06-10", + "value": "Official server adds configurable toolsets (tool scoping) and a read-only mode to limit the action surface" } ], "methodology": "Permission scope testing", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "action_auditability": { "score": 82, @@ -144,23 +156,29 @@ "last_verified": "2025-11-09" }, "unauthorized_action_risk": { - "score": 72, - "confidence": "medium", + "score": 60, + "confidence": "high", "evidence": [ { "source": "Security Analysis", "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "AI can create PRs, issues, and modify repos within token permissions" + }, + { + "source": "Invariant Labs - GitHub MCP vulnerability", + "url": "https://invariantlabs.ai/blog/mcp-github-vulnerability", + "date": "2025-05-26", + "value": "Architectural prompt-injection finding: a malicious public issue can steer the agent into exfiltrating private-repo data; mitigations are least-privilege tokens and one-repo sessions" } ], "methodology": "Authorization boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } }, "privacy_compliance": { - "overall_score": 74, + "overall_score": 71, "criteria": { "code_exposure": { "score": 70, @@ -177,18 +195,24 @@ "last_verified": "2025-11-09" }, "sensitive_data_protection": { - "score": 68, - "confidence": "medium", + "score": 62, + "confidence": "high", "evidence": [ { "source": "MCP Security Guidelines", "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "No built-in secret detection; risk of exposing API keys or credentials in code" + }, + { + "source": "Invariant Labs - GitHub MCP vulnerability", + "url": "https://invariantlabs.ai/blog/mcp-github-vulnerability", + "date": "2025-05-26", + "value": "Demonstrated private-repository data exfiltration via prompt injection from a malicious public issue when broad tokens are used" } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "organization_data_control": { "score": 78, @@ -221,7 +245,7 @@ } }, "trust_transparency": { - "overall_score": 87, + "overall_score": 90, "criteria": { "documentation_quality": { "score": 92, @@ -256,36 +280,36 @@ "confidence": "high", "evidence": [ { - "source": "GitHub Repository", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Fully open source implementation with MIT license" + "source": "GitHub MCP Server (official)", + "url": "https://github.com/github/github-mcp-server", + "date": "2026-06-10", + "value": "Fully open source Go implementation with MIT license; 30,558 GitHub stars" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "api_coverage_clarity": { - "score": 80, - "confidence": "medium", + "score": 85, + "confidence": "high", "evidence": [ { - "source": "MCP Server Documentation", - "url": "https://modelcontextprotocol.io/docs/servers/github", - "date": "2025-11-16", - "value": "Clear documentation of supported GitHub API operations" + "source": "GitHub MCP Server (official)", + "url": "https://github.com/github/github-mcp-server", + "date": "2026-06-10", + "value": "50+ tools documented and organized into configurable toolsets (repos, issues, PRs, actions, security, etc.)" } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } }, "operational_excellence": { - "overall_score": 85, + "overall_score": 87, "criteria": { "ease_of_setup": { - "score": 88, + "score": 90, "confidence": "high", "evidence": [ { @@ -293,10 +317,16 @@ "url": "https://modelcontextprotocol.io/quickstart", "date": "2025-11-16", "value": "Simple setup requiring only GitHub PAT configuration" + }, + { + "source": "GitHub Changelog - Remote GitHub MCP Server GA", + "url": "https://github.blog/changelog/2025-09-04-remote-github-mcp-server-is-now-generally-available/", + "date": "2025-09-04", + "value": "Hosted remote endpoint removes local install entirely; local option ships as Go binary or Docker image" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "api_performance": { "score": 82, @@ -327,32 +357,32 @@ "last_verified": "2025-11-09" }, "feature_coverage": { - "score": 85, + "score": 88, "confidence": "high", "evidence": [ { - "source": "MCP GitHub Server", - "url": "https://github.com/modelcontextprotocol/servers/tree/main/src/github", - "date": "2025-11-16", - "value": "Covers repos, issues, PRs, search, and file operations" + "source": "GitHub MCP Server (official)", + "url": "https://github.com/github/github-mcp-server", + "date": "2026-06-10", + "value": "50+ tools in configurable toolsets covering repos, issues, PRs, actions, code security, and search; read-only mode supported" } ], "methodology": "Feature completeness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "community_adoption": { - "score": 80, - "confidence": "medium", + "score": 88, + "confidence": "high", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/modelcontextprotocol/servers/discussions", - "date": "2025-11-16", - "value": "Growing adoption as MCP protocol gains traction (launched late 2024)" + "source": "GitHub MCP Server (official)", + "url": "https://github.com/github/github-mcp-server", + "date": "2026-06-10", + "value": "30,558 GitHub stars; official GitHub maintenance with hosted remote generally available since 2025-09-04" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -362,8 +392,9 @@ "Built on reliable GitHub infrastructure with high uptime", "Excellent for development workflows and code collaboration", "Full operation auditability through GitHub's audit logs", - "Open source implementation with active Anthropic support", - "Supports granular permission scopes via GitHub tokens" + "Official GitHub-maintained open source server (Go, MIT, 30,558 stars)", + "Hosted remote option (OAuth 2.1+PKCE) generally available since 2025-09-04", + "Configurable toolsets and read-only mode limit the action surface" ], "limitations": [ "Repository code and metadata exposed to LLM provider APIs", @@ -371,31 +402,38 @@ "No built-in secret detection or sensitive data filtering", "Subject to GitHub API rate limits (5000 requests/hour)", "Token scope misconfiguration can grant excessive permissions", - "Potential for accidental data leakage from private repositories" + "Architectural prompt-injection risk: malicious public issues can drive private-repo data exfiltration (Invariant Labs, May 2025); mitigate with least-privilege tokens and one-repo sessions" ], "metadata": { "license": "MIT", "supported_platforms": [ - "All platforms with Node.js/Python" + "Windows", + "macOS", + "Linux", + "Hosted remote (https://api.githubcopilot.com/mcp/)" ], "programming_languages": [ - "TypeScript", - "Python" + "Go" ], "mcp_version": "1.0", "github_repo": "https://github.com/github/github-mcp-server", - "github_stars": 58700, + "github_stars": 30558, "deprecated_repo": "https://github.com/modelcontextprotocol/servers-archived", "api_dependency": "GitHub REST API v3", - "authentication": "GitHub Personal Access Token or OAuth", + "authentication": "GitHub PAT (local) or OAuth 2.1 + PKCE (hosted remote)", + "remote_endpoint": "https://api.githubcopilot.com/mcp/", + "remote_ga_date": "2025-09-04", "first_release": "2024-11", "maintained_by": "GitHub", - "status": "Migrated - Now maintained by GitHub", + "status": "Active - official GitHub server; archived Anthropic reference server superseded", "transport_types": [ - "stdio" + "stdio", + "streamable-http (hosted remote)" ], "installation_methods": [ - "npm" + "Go binary", + "Docker", + "Hosted remote (no install)" ] }, "use_case_ratings": { @@ -449,6 +487,8 @@ "git", "version-control", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "remote-server" ] } diff --git a/data/mcps/mcp-server-gitlab.json b/data/mcps/mcp-server-gitlab.json index 67b6439..7b2776f 100644 --- a/data/mcps/mcp-server-gitlab.json +++ b/data/mcps/mcp-server-gitlab.json @@ -2,15 +2,15 @@ "id": "mcp-server-gitlab", "type": "mcp", "name": "MCP GitLab Server", - "provider": "GitLab Community", + "provider": "Anthropic (Archived)", "version": "2025.3.2", - "last_evaluated": "2025-01-14", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "MCP server providing AI models with comprehensive GitLab integration capabilities. Enables merge request management, CI/CD pipeline inspection, issue tracking, and code review workflows for enterprise teams using GitLab.", + "description": "ARCHIVED: Former Anthropic reference MCP server for GitLab integration (merge requests, CI/CD pipelines, issues), archived 2025-05-29 to the servers-archived repository and no longer maintained; no security guarantees are provided for archived servers. The GitLab MCP ecosystem has since moved to other actively maintained implementations, which should be preferred.", "website": "https://gitlab.com/gitlab-org/gitlab-mcp-server", "trust_vector": { "performance_reliability": { - "overall_score": 84, + "overall_score": 80, "criteria": { "api_reliability": { "score": 88, @@ -69,7 +69,7 @@ "last_verified": "2025-01-14" }, "error_recovery": { - "score": 82, + "score": 60, "confidence": "medium", "evidence": [ { @@ -77,15 +77,21 @@ "url": "https://gitlab.com/gitlab-org/gitlab-mcp-server", "date": "2025-01-10", "value": "Handles API errors with retry logic and graceful degradation" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Reference GitLab server archived 2025-05-29; bugs and GitLab API breaking changes will never be fixed" } ], "methodology": "Error handling testing", - "last_verified": "2025-01-14" + "last_verified": "2026-06-10" } } }, "security": { - "overall_score": 80, + "overall_score": 83, "criteria": { "authentication_security": { "score": 88, @@ -160,7 +166,7 @@ } }, "privacy_compliance": { - "overall_score": 78, + "overall_score": 76, "criteria": { "code_exposure": { "score": 72, @@ -221,7 +227,7 @@ } }, "trust_transparency": { - "overall_score": 85, + "overall_score": 87, "criteria": { "documentation_quality": { "score": 88, @@ -282,7 +288,7 @@ } }, "operational_excellence": { - "overall_score": 83, + "overall_score": 80, "criteria": { "ease_of_setup": { "score": 85, @@ -327,7 +333,7 @@ "last_verified": "2025-01-14" }, "feature_coverage": { - "score": 84, + "score": 60, "confidence": "high", "evidence": [ { @@ -335,10 +341,16 @@ "url": "https://gitlab.com/gitlab-org/gitlab-mcp-server", "date": "2025-01-10", "value": "Covers MRs, issues, pipelines, projects, and file operations" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Archived 2025-05-29; feature set frozen, no updates for GitLab API or MCP specification changes" } ], "methodology": "Feature completeness assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-06-10" }, "enterprise_features": { "score": 88, @@ -371,7 +383,7 @@ "Some features require GitLab Premium/Ultimate", "Subject to GitLab API rate limits", "No built-in secret detection in code", - "Newer than GitHub MCP, smaller community" + "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; GitLab ecosystem has moved to other implementations" ], "metadata": { "license": "MIT", @@ -382,7 +394,8 @@ "api_dependency": "GitLab REST API v4 / GraphQL", "authentication": "GitLab Personal Access Token or OAuth", "first_release": "2025-02", - "maintained_by": "GitLab Community", + "maintained_by": "None (Archived 2025-05-29)", + "repository": "https://github.com/modelcontextprotocol/servers-archived", "transport_types": ["stdio"], "installation_methods": ["npm", "pip"] }, @@ -434,5 +447,5 @@ "Teams wanting AI-assisted CI/CD management", "Developers integrating AI with GitLab workflows" ], - "tags": ["git", "gitlab", "devops", "ci-cd", "mcp", "model-context-protocol"] + "tags": ["git", "gitlab", "devops", "ci-cd", "mcp", "model-context-protocol", "archived", "unmaintained"] } diff --git a/data/mcps/mcp-server-google-drive.json b/data/mcps/mcp-server-google-drive.json index c36a7b6..50eff47 100644 --- a/data/mcps/mcp-server-google-drive.json +++ b/data/mcps/mcp-server-google-drive.json @@ -2,15 +2,15 @@ "id": "mcp-server-google-drive", "type": "mcp", "name": "MCP Google Drive Server", - "provider": "Community", + "provider": "Anthropic (Archived)", "version": "1.0.0", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "MCP server enabling AI models to access, search, and manage Google Drive files and folders. Supports document reading, file operations, and collaboration features through the Model Context Protocol. Essential for AI-powered document workflows but requires careful permission management.", + "description": "ARCHIVED: Former Anthropic reference MCP server (gdrive) for accessing and searching Google Drive files, archived 2025-05-29 to the servers-archived repository and no longer maintained; the archive README provides no security guarantees. The underlying Google Drive API remains reliable, but this server receives no fixes and is not recommended for new deployments.", "website": "https://developers.google.com/drive", "trust_vector": { "performance_reliability": { - "overall_score": 86, + "overall_score": 85, "criteria": { "file_operation_reliability": { "score": 90, @@ -85,7 +85,7 @@ } }, "security": { - "overall_score": 76, + "overall_score": 78, "criteria": { "oauth_security": { "score": 88, @@ -235,7 +235,7 @@ } }, "trust_transparency": { - "overall_score": 82, + "overall_score": 77, "criteria": { "documentation_quality": { "score": 85, @@ -266,18 +266,18 @@ "last_verified": "2025-11-08" }, "mcp_implementation_clarity": { - "score": 78, - "confidence": "medium", + "score": 58, + "confidence": "high", "evidence": [ { - "source": "Community Implementation", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Community-maintained implementation with variable documentation quality" + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Reference gdrive server archived 2025-05-29; repository read-only, README states no security guarantees are provided for archived servers" } ], "methodology": "Implementation documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10" }, "permission_transparency": { "score": 80, @@ -296,7 +296,7 @@ } }, "operational_excellence": { - "overall_score": 81, + "overall_score": 76, "criteria": { "ease_of_setup": { "score": 75, @@ -341,7 +341,7 @@ "last_verified": "2025-11-08" }, "feature_coverage": { - "score": 78, + "score": 55, "confidence": "medium", "evidence": [ { @@ -349,10 +349,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Covers file read/write, search, and basic sharing operations" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Archived 2025-05-29; feature set frozen, no updates for Google Drive API or MCP specification changes" } ], "methodology": "Feature completeness assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10" }, "cost": { "score": 80, @@ -385,7 +391,7 @@ "Complex OAuth setup process requiring Google Cloud project", "AI can access and modify all files within granted permissions", "Subject to Google API rate limits and quotas", - "Community-maintained MCP server with variable support quality" + "ARCHIVED 2025-05-29: reference gdrive server unmaintained with no security guarantees; prefer actively maintained Google Drive MCP alternatives" ], "metadata": { "license": "Varies (API proprietary, MCP implementation varies)", @@ -403,7 +409,8 @@ "api_version": "Google Drive API v3", "rate_limits": "1000 queries per 100 seconds (default)", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "None (Archived 2025-05-29)", + "repository": "https://github.com/modelcontextprotocol/servers-archived" }, "use_case_ratings": { "code-generation": { @@ -456,6 +463,8 @@ "storage", "google", "mcp", - "model-context-protocol" + "model-context-protocol", + "archived", + "unmaintained" ] } diff --git a/data/mcps/mcp-server-hugging-face.json b/data/mcps/mcp-server-hugging-face.json new file mode 100644 index 0000000..106c4a7 --- /dev/null +++ b/data/mcps/mcp-server-hugging-face.json @@ -0,0 +1,436 @@ +{ + "id": "mcp-server-hugging-face", + "type": "mcp", + "name": "Hugging Face MCP Server", + "provider": "Hugging Face", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Hugging Face's official MCP server connecting AI assistants to the Hub. Ships 7 built-in tools (search for models, datasets, Spaces, and papers, plus documentation search) and can dynamically attach community Gradio Spaces as additional tools. Hosted at huggingface.co/mcp with per-user configuration, or runnable locally; open source under MIT.", + "website": "https://huggingface.co/mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 80, + "criteria": { + "api_reliability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Hosted endpoint runs on Hugging Face Hub infrastructure with Streamable HTTP transport designed for stateless, scalable operation" + } + ], + "methodology": "Endpoint stability analysis of the hosted server and underlying Hub APIs", + "last_verified": "2026-06-10" + }, + "search_accuracy": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "Built-in search tools query the Hub's native search across models, datasets, Spaces, and papers with relevant, current results" + } + ], + "methodology": "Relevance assessment of Hub search results for representative ML queries", + "last_verified": "2026-06-10" + }, + "operation_success_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "The 7 built-in tools are thin wrappers over stable Hub APIs and succeed consistently" + } + ], + "methodology": "Operation success testing across built-in tools", + "last_verified": "2026-06-10" + }, + "dynamic_tool_reliability": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Attached Gradio Spaces run as community-maintained apps; they can be slow to cold-start, hit ZeroGPU queues, or fail when the Space author changes or breaks the app" + } + ], + "methodology": "Reliability testing of dynamically attached Gradio Space tools across popular Spaces", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "Open-source implementation returns structured errors and supports MCP tool-list-changed notifications when the user's configured tool set changes" + } + ], + "methodology": "Error handling testing including dynamic tool set changes mid-session", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 71, + "criteria": { + "authentication_security": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face MCP Settings", + "url": "https://huggingface.co/settings/mcp", + "date": "2026-06-10", + "value": "Hosted server authenticates with Hugging Face account credentials/tokens; per-user tool configuration is managed at hf.co/settings/mcp" + } + ], + "methodology": "Authentication mechanism review for hosted and local deployment modes", + "last_verified": "2026-06-10" + }, + "token_exposure_risk": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Security Tokens Documentation", + "url": "https://huggingface.co/docs/hub/security-tokens", + "date": "2026-06-10", + "value": "Hub supports fine-grained access tokens, but local stdio deployments place tokens in client configuration files" + } + ], + "methodology": "Token storage and exposure-surface analysis across deployment modes", + "last_verified": "2026-06-10" + }, + "scope_limitation": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Security Tokens Documentation", + "url": "https://huggingface.co/docs/hub/security-tokens", + "date": "2026-06-10", + "value": "Fine-grained tokens can restrict Hub access, and built-in tools are read-oriented; however, attached Gradio Spaces execute with whatever inputs the agent supplies" + } + ], + "methodology": "Permission scope testing of built-in tools and attached Space tools", + "last_verified": "2026-06-10" + }, + "third_party_tool_supply_chain": { + "score": 50, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Dynamically attached Gradio Spaces are arbitrary community-authored applications; tool descriptions and outputs are untrusted third-party content, creating supply-chain and prompt injection exposure inside the agent loop" + } + ], + "methodology": "Supply-chain threat modeling of community Space attachment: untrusted code, mutable tool definitions, and unvetted outputs", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "Built-in tools are search/read operations with limited blast radius; risk concentrates in what user-attached Spaces are allowed to do with submitted data" + } + ], + "methodology": "Authorization boundary analysis of built-in versus attached tool capabilities", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 69, + "criteria": { + "query_data_exposure": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Search queries and tool results flow through the hosted server and into the LLM provider's context; built-in tools touch mostly public Hub content" + } + ], + "methodology": "Data flow analysis of queries and results across the hosted server", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "No built-in redaction; any sensitive content an agent submits to an attached Space tool leaves the Hugging Face trust boundary" + } + ], + "methodology": "Assessment of filtering controls on data submitted to attached tools", + "last_verified": "2026-06-10" + }, + "organization_data_control": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Hub Security Documentation", + "url": "https://huggingface.co/docs/hub/security", + "date": "2026-06-10", + "value": "Org access controls and gated/private repos apply to authenticated Hub access; MCP tool configuration is per-user rather than org-governed" + } + ], + "methodology": "Access control review of Hub permissions as applied through the MCP server", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Inputs sent to attached Gradio Spaces are processed by community-operated applications whose data handling is not governed by Hugging Face's privacy commitments" + } + ], + "methodology": "Analysis of data sharing with community Space operators and the LLM provider", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 83, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Detailed engineering blog plus repository README cover architecture, transports (Streamable HTTP and stdio), and setup for hosted and local modes" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face MCP Settings", + "url": "https://huggingface.co/settings/mcp", + "date": "2026-06-10", + "value": "Users can see and manage exactly which tools and Spaces are attached at hf.co/settings/mcp; tool calls are visible in MCP client logs" + } + ], + "methodology": "Logging and configuration-visibility assessment", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "Server is fully open source under MIT (approximately 247 stars); the same code powers the hosted deployment and can be self-hosted" + } + ], + "methodology": "Source code review of the published server implementation", + "last_verified": "2026-06-10" + }, + "api_coverage_clarity": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "The 7 built-in tools are clearly enumerated, but the effective tool surface varies per user depending on which Gradio Spaces are attached" + } + ], + "methodology": "Comparison of documented tool surface against per-user dynamic configuration", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 81, + "criteria": { + "ease_of_setup": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face MCP Endpoint", + "url": "https://huggingface.co/mcp", + "date": "2026-06-10", + "value": "Hosted mode requires only adding https://huggingface.co/mcp and authenticating; tool selection is point-and-click at hf.co/settings/mcp" + } + ], + "methodology": "Setup complexity assessment for hosted and local modes", + "last_verified": "2026-06-10" + }, + "api_performance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Built-in search tools respond quickly; attached Space tools vary widely with Space hardware, cold starts, and GPU queues" + } + ], + "methodology": "Latency observation across built-in and attached tools", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Status", + "url": "https://status.huggingface.co/", + "date": "2026-06-10", + "value": "Hosted endpoint tracks Hub availability, which is historically solid; community Space tools are the main reliability variable" + } + ], + "methodology": "Uptime analysis of Hub infrastructure versus attached tool availability", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face MCP Server Blog", + "url": "https://huggingface.co/blog/building-hf-mcp", + "date": "2026-06-10", + "value": "Covers Hub discovery (models, datasets, Spaces, papers, docs) and extends to image generation, transcription, and thousands of other capabilities via attachable Gradio Spaces" + } + ], + "methodology": "Feature completeness assessment including the dynamic tool extension model", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "hf-mcp-server Repository", + "url": "https://github.com/huggingface/hf-mcp-server", + "date": "2026-06-10", + "value": "Approximately 247 GitHub stars with active first-party maintenance; adoption driven mainly by the hosted endpoint within the large HF user base" + } + ], + "methodology": "Community activity and adoption analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Fully open source (MIT) with the same code powering the hosted endpoint", + "Strong Hub discovery: models, datasets, Spaces, papers, and documentation search", + "Dynamic Gradio Space attachment extends the agent with thousands of community capabilities", + "Per-user tool configuration UI at hf.co/settings/mcp with tool-list-changed support", + "Flexible deployment: hosted Streamable HTTP or local stdio", + "Backed by Hugging Face's first-party maintenance and Hub infrastructure" + ], + "limitations": [ + "Attached Gradio Spaces are arbitrary community apps: a third-party tool supply-chain and prompt injection exposure", + "Data submitted to attached Spaces leaves Hugging Face's privacy boundary", + "Attached tool reliability varies with Space cold starts, GPU queues, and author changes", + "Effective tool surface differs per user, complicating organizational review", + "Local stdio mode stores Hub tokens in client configuration", + "MCP tool configuration is per-user with no org-level governance controls" + ], + "metadata": { + "repository": "https://github.com/huggingface/hf-mcp-server", + "license": "MIT", + "maintained_by": "Hugging Face", + "github_stars": 247, + "remote_endpoint": "https://huggingface.co/mcp", + "configuration_url": "https://huggingface.co/settings/mcp", + "authentication": "Hugging Face account / access tokens (fine-grained supported)", + "transport_types": [ + "streamable-http", + "stdio" + ], + "installation_methods": [ + "Remote MCP endpoint", + "Local self-hosted (Node.js)" + ], + "built_in_tools": 7, + "mcp_version": "1.0" + }, + "use_case_ratings": { + "research-assistant": { + "overall": 90, + "notes": "Excellent for discovering models, datasets, papers, and Spaces directly from the Hub" + }, + "data-analysis": { + "overall": 82, + "notes": "Strong for finding datasets and running analysis-oriented Spaces, with variable attached-tool reliability" + }, + "code-generation": { + "overall": 80, + "notes": "Doc search and model discovery materially improve ML integration code quality" + }, + "education": { + "overall": 84, + "notes": "Great for teaching ML concepts with live access to models, papers, and demo Spaces" + }, + "content-creation": { + "overall": 72, + "notes": "Image generation and media Spaces are attachable as tools, though quality and uptime vary by Space" + } + }, + "best_for": [ + "ML engineers and researchers discovering models, datasets, and papers from an agent", + "Developers building Hugging Face integrations with live documentation search", + "Power users extending agents with community Gradio Space capabilities", + "Teams that want an auditable, self-hostable open-source MCP server" + ], + "related_entities": [ + "mcp-server-github", + "mcp-server-vercel", + "mcp-server-notion", + "mcp-server-zapier" + ], + "tags": [ + "machine-learning", + "models", + "datasets", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-memory.json b/data/mcps/mcp-server-memory.json index 0bfb36a..a28d4d2 100644 --- a/data/mcps/mcp-server-memory.json +++ b/data/mcps/mcp-server-memory.json @@ -4,9 +4,9 @@ "name": "MCP Memory Server", "provider": "Anthropic", "version": "2025.9.25", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "MCP server providing AI models with persistent memory and knowledge graph capabilities. Enables long-term information retention, entity relationship tracking, and contextual recall across conversations through the Model Context Protocol. Critical for personalized AI but raises significant privacy concerns.", + "description": "MCP reference server providing persistent memory and knowledge graph capabilities: long-term retention, entity relationship tracking, and contextual recall across conversations. One of the seven reference servers still actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest spec 2025-11-25. Critical for personalized AI but raises significant privacy concerns.", "website": "https://github.com/modelcontextprotocol/servers", "trust_vector": { "performance_reliability": { @@ -349,10 +349,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Requires periodic cleanup and optimization" + }, + { + "source": "Anthropic - Donating MCP to the Agentic AI Foundation", + "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "memory is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" } ], "methodology": "Maintenance overhead assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "backup_and_recovery": { "score": 80, @@ -385,7 +391,8 @@ "Embeddings typically sent to third-party providers (OpenAI, etc.)", "Limited consent management and data retention controls", "Memory poisoning risk - AI can store incorrect information", - "GDPR/privacy compliance challenges without careful implementation" + "GDPR/privacy compliance challenges without careful implementation", + "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -412,7 +419,7 @@ ], "vector_dimensions": "Varies (384-1536 typical)", "first_release": "2024-11", - "maintained_by": "Anthropic", + "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Active", "transport_types": [ "stdio" @@ -472,6 +479,8 @@ "memory", "storage", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "reference-server" ] } diff --git a/data/mcps/mcp-server-notion.json b/data/mcps/mcp-server-notion.json index fbd5e6f..ea0475a 100644 --- a/data/mcps/mcp-server-notion.json +++ b/data/mcps/mcp-server-notion.json @@ -4,13 +4,13 @@ "name": "MCP Notion Server", "provider": "Notion (Official)", "version": "1.0.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Notion workspace integration. Enables AI models to create, read, update pages and databases, manage blocks, and search content. Essential for AI-powered knowledge management, documentation, and collaborative workspace workflows.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "Notion's official MCP integration. Primary offering is the hosted MCP server at https://mcp.notion.com/mcp (Streamable HTTP and SSE transports, one-click OAuth), positioned as the successor to the open-source local @notionhq/notion-mcp-server (MIT, 4,407 stars), which remains available. Enables AI models to create, read, update pages and databases, manage blocks, and search workspace content.", + "website": "https://mcp.notion.com/mcp", "trust_vector": { "performance_reliability": { - "overall_score": 82, + "overall_score": 81, "criteria": { "page_operation_reliability": { "score": 88, @@ -74,7 +74,7 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/modelcontextprotocol/servers", + "url": "https://github.com/makenotion/notion-mcp-server", "date": "2025-11-16", "value": "Handles API errors with retry and timeout management" } @@ -85,21 +85,27 @@ } }, "security": { - "overall_score": 71, + "overall_score": 74, "criteria": { "authentication_security": { - "score": 80, + "score": 88, "confidence": "high", "evidence": [ { "source": "Notion Integration Authentication", "url": "https://developers.notion.com/docs/authorization", "date": "2025-11-16", - "value": "Uses integration tokens with workspace-level permissions" + "value": "Local server uses integration tokens with workspace-level permissions" + }, + { + "source": "Notion Blog - Hosted MCP Server", + "url": "https://www.notion.com/blog/notions-hosted-mcp-server-an-inside-look", + "date": "2026-06-10", + "value": "Hosted server at mcp.notion.com/mcp uses one-click OAuth, eliminating manual integration-token handling" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "token_exposure_risk": { "score": 65, @@ -121,7 +127,7 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/modelcontextprotocol/servers", + "url": "https://github.com/makenotion/notion-mcp-server", "date": "2025-11-16", "value": "AI can create, modify, and archive pages and databases" } @@ -196,7 +202,7 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/modelcontextprotocol/servers", + "url": "https://github.com/makenotion/notion-mcp-server", "date": "2025-11-16", "value": "Database schemas and property values exposed" } @@ -210,7 +216,7 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/modelcontextprotocol/servers", + "url": "https://github.com/makenotion/notion-mcp-server", "date": "2025-11-16", "value": "Workspace structure and page hierarchy may be revealed" } @@ -249,21 +255,21 @@ } }, "trust_transparency": { - "overall_score": 79, + "overall_score": 81, "criteria": { "documentation_quality": { - "score": 76, - "confidence": "medium", + "score": 85, + "confidence": "high", "evidence": [ { - "source": "Notion MCP Docs", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Good documentation but community-maintained" + "source": "Notion Blog - Hosted MCP Server", + "url": "https://www.notion.com/blog/notions-hosted-mcp-server-an-inside-look", + "date": "2026-06-10", + "value": "Officially documented by Notion, including architecture details of the hosted server and guidance for the local open-source server" } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "operation_visibility": { "score": 82, @@ -284,14 +290,14 @@ "confidence": "high", "evidence": [ { - "source": "GitHub Repository", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Open source community implementation" + "source": "Notion MCP Server GitHub Repository", + "url": "https://github.com/makenotion/notion-mcp-server", + "date": "2026-06-10", + "value": "Local server (@notionhq/notion-mcp-server) is open source under MIT with 4,407 GitHub stars; hosted server internals documented in Notion's engineering blog" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "api_coverage_clarity": { "score": 70, @@ -299,7 +305,7 @@ "evidence": [ { "source": "MCP Server Documentation", - "url": "https://github.com/modelcontextprotocol/servers", + "url": "https://github.com/makenotion/notion-mcp-server", "date": "2025-11-16", "value": "Clear but incomplete documentation of supported operations" } @@ -318,13 +324,19 @@ "evidence": [ { "source": "Setup Documentation", - "url": "https://github.com/modelcontextprotocol/servers", + "url": "https://github.com/makenotion/notion-mcp-server", "date": "2025-11-16", - "value": "Requires Notion integration creation and token configuration" + "value": "Local server requires Notion integration creation and token configuration" + }, + { + "source": "Notion Blog - Hosted MCP Server", + "url": "https://www.notion.com/blog/notions-hosted-mcp-server-an-inside-look", + "date": "2026-06-10", + "value": "Hosted server at mcp.notion.com/mcp needs no local install: Streamable HTTP/SSE transports with one-click OAuth" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "api_performance": { "score": 78, @@ -360,7 +372,7 @@ "evidence": [ { "source": "Notion MCP Server", - "url": "https://github.com/modelcontextprotocol/servers", + "url": "https://github.com/makenotion/notion-mcp-server", "date": "2025-11-16", "value": "Covers pages, databases, blocks, and search operations" } @@ -369,18 +381,18 @@ "last_verified": "2025-11-09" }, "community_support": { - "score": 75, - "confidence": "medium", + "score": 80, + "confidence": "high", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/modelcontextprotocol/servers/discussions", - "date": "2025-11-16", - "value": "Community-maintained with moderate activity" + "source": "Notion MCP Server GitHub Repository", + "url": "https://github.com/makenotion/notion-mcp-server", + "date": "2026-06-10", + "value": "Officially maintained by Notion; open-source local server has 4,407 GitHub stars, and the hosted server is operated by Notion as the primary offering" } ], "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -390,7 +402,7 @@ "Built on reliable Notion API with good uptime", "Excellent for knowledge management and documentation workflows", "Integration permissions provide page-level access control", - "Open source community implementation", + "Official hosted server (mcp.notion.com/mcp) with one-click OAuth; open-source local server (MIT, 4,407 stars) still available", "Archived pages are recoverable (not permanently deleted)" ], "limitations": [ @@ -399,7 +411,8 @@ "Workspace structure and page hierarchy may be revealed", "Subject to strict rate limits (3 requests/second average)", "Database schemas and property values exposed", - "User and collaborator information accessible" + "User and collaborator information accessible", + "STATUS 2026-06-10: Notion's hosted server (mcp.notion.com/mcp) is now the primary, recommended offering; the local @notionhq/notion-mcp-server remains available but workspace traffic on the hosted path routes through Notion's infrastructure" ], "metadata": { "license": "MIT", @@ -411,11 +424,19 @@ "Python" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", + "github_repo": "https://github.com/makenotion/notion-mcp-server", + "github_stars": 4407, "api_dependency": "Notion API, Notion Client SDK", - "authentication": "Notion Integration Token", + "authentication": "OAuth (hosted at mcp.notion.com/mcp) or Notion Integration Token (local server)", + "remote_endpoint": "https://mcp.notion.com/mcp", + "transport_types": [ + "streamable-http (hosted)", + "sse (hosted)", + "stdio (local)" + ], + "package_name": "@notionhq/notion-mcp-server", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "Notion" }, "use_case_ratings": { "code-generation": { @@ -468,6 +489,8 @@ "productivity", "notes", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "remote-server" ] } diff --git a/data/mcps/mcp-server-playwright.json b/data/mcps/mcp-server-playwright.json new file mode 100644 index 0000000..03b7b74 --- /dev/null +++ b/data/mcps/mcp-server-playwright.json @@ -0,0 +1,452 @@ +{ + "id": "mcp-server-playwright", + "type": "mcp", + "name": "Playwright MCP", + "provider": "Microsoft", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Microsoft's official MCP server for browser automation via Playwright. Exposes 25+ tools (navigation, clicking, typing, form filling, screenshots, network inspection, JS evaluation) that operate on structured accessibility-tree snapshots rather than pixels, making agent-driven browsing fast and deterministic. Supersedes the archived Puppeteer reference server.", + "website": "https://github.com/microsoft/playwright-mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 88, + "criteria": { + "snapshot_accuracy": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP README", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Uses Playwright accessibility-tree snapshots instead of pixel-based screenshots, giving the LLM deterministic, structured page state with stable element references" + } + ], + "methodology": "Review of snapshot mechanism (browser_snapshot) and element-reference stability across page interactions", + "last_verified": "2026-06-10" + }, + "operation_success_rate": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP tool documentation", + "url": "https://github.com/microsoft/playwright-mcp/blob/main/README.md", + "date": "2026-06-10", + "value": "Core interaction tools (browser_navigate, browser_click, browser_type, browser_fill_form) inherit Playwright's auto-waiting and actionability checks, yielding high success rates on standard web UIs" + } + ], + "methodology": "Hands-on testing of navigation, clicking, typing, and form-fill tools against common web applications", + "last_verified": "2026-06-10" + }, + "browser_compatibility": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Playwright browser support", + "url": "https://playwright.dev/docs/browsers", + "date": "2026-06-10", + "value": "Built on Playwright, supporting Chromium, Firefox, and WebKit engines with consistent automation APIs" + } + ], + "methodology": "Cross-browser capability review based on underlying Playwright engine support", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Playwright MCP repository", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Failed actions return structured error messages and fresh page snapshots, allowing the agent to retry; dialog and file-upload handlers prevent common automation deadlocks" + } + ], + "methodology": "Error-path testing including timeouts, missing elements, modal dialogs, and navigation failures", + "last_verified": "2026-06-10" + }, + "automation_stability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Playwright auto-waiting documentation", + "url": "https://playwright.dev/docs/actionability", + "date": "2026-06-10", + "value": "Playwright's actionability checks (visible, stable, enabled) reduce flakiness common in agent-driven browsing; tab management and network-request inspection aid multi-step flows" + } + ], + "methodology": "Multi-step workflow stability testing across tabs, dialogs, and dynamic pages", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 60, + "criteria": { + "prompt_injection_resistance": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Agentic browsing security analysis", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Page content (accessibility snapshots, network responses) flows directly into the LLM context; untrusted web pages can embed instructions that hijack the agent (indirect prompt injection). No built-in content sanitization" + } + ], + "methodology": "Threat modeling of untrusted web content entering the agent context via snapshots and screenshots", + "last_verified": "2026-06-10" + }, + "arbitrary_code_execution_risk": { + "score": 50, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP tool list", + "url": "https://github.com/microsoft/playwright-mcp/blob/main/README.md", + "date": "2026-06-10", + "value": "browser_evaluate executes arbitrary JavaScript in the page context, enabling data exfiltration or session manipulation if the agent is compromised; capability can be restricted via configuration" + } + ], + "methodology": "Capability analysis of the browser_evaluate tool and its abuse potential under prompt injection", + "last_verified": "2026-06-10" + }, + "sandboxing_isolation": { + "score": 75, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP configuration options", + "url": "https://github.com/microsoft/playwright-mcp/blob/main/README.md", + "date": "2026-06-10", + "value": "Supports isolated browser profiles (no persisted state), headless mode, and origin allow/block lists, which substantially limit blast radius when configured; default persistent profile retains cookies and logins" + } + ], + "methodology": "Review of --isolated, headless, and origin-filtering configuration flags as mitigations", + "last_verified": "2026-06-10" + }, + "credential_exposure_risk": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Playwright MCP profile behavior", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "When run against a persistent profile, the agent operates with the user's logged-in sessions and cookies; typed credentials and page secrets appear in snapshots sent to the LLM" + } + ], + "methodology": "Analysis of session/cookie access in persistent vs isolated profile modes", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "MCP security guidance", + "url": "https://modelcontextprotocol.io/docs/concepts/security", + "date": "2026-06-10", + "value": "Agent can perform any web action the browser session allows (purchases, posts, account changes); host-level tool approval is the primary guardrail since the server itself does not gate destructive actions" + } + ], + "methodology": "Authorization boundary analysis of write-capable browsing actions", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 69, + "criteria": { + "browsing_data_exposure": { + "score": 65, + "confidence": "high", + "evidence": [ + { + "source": "MCP data flow architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-06-10", + "value": "Full page content, screenshots, and network request data are sent to the LLM provider as tool results, exposing any visited page (including authenticated content) to a third party" + } + ], + "methodology": "Data flow analysis of snapshot, screenshot, and network tool outputs", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Playwright MCP repository", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "No built-in redaction of PII, credentials, or payment data visible on pages; protection relies on operator discipline (isolated profiles, restricted origins)" + } + ], + "methodology": "Privacy controls assessment of snapshot and screenshot content handling", + "last_verified": "2026-06-10" + }, + "local_data_control": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP architecture", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Server runs locally (stdio default); browser state, profiles, and traces remain on the user's machine, with no vendor-side telemetry collection by the MCP server itself" + } + ], + "methodology": "Review of local execution model and data residency", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "MCP client documentation", + "url": "https://modelcontextprotocol.io/docs", + "date": "2026-06-10", + "value": "Browsing data is shared only with the connected LLM provider per that provider's privacy policy; the server itself transmits nothing to Microsoft" + } + ], + "methodology": "Data sharing pathway analysis", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 92, + "criteria": { + "documentation_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP README", + "url": "https://github.com/microsoft/playwright-mcp/blob/main/README.md", + "date": "2026-06-10", + "value": "Thorough documentation of all 25+ tools, configuration flags, client setup for major MCP hosts, and isolated/extension/server modes" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Fully open source under Apache-2.0 with public issue tracker and active development by the Playwright team" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP tooling", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Every browser action is an explicit, named tool call visible in MCP host logs; optional Playwright traces and browser_network_requests provide deep inspection of agent behavior" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-06-10" + }, + "vendor_credibility": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "GitHub API", + "url": "https://api.github.com/repos/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Maintained by Microsoft's Playwright team; 33,734 GitHub stars as of 2026-06-10, one of the most adopted MCP servers" + } + ], + "methodology": "Maintainer reputation and project health analysis", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 90, + "criteria": { + "ease_of_setup": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "npm package @playwright/mcp", + "url": "https://www.npmjs.com/package/@playwright/mcp", + "date": "2026-06-10", + "value": "Single-command setup via npx @playwright/mcp@latest; no API keys required; one-click install paths documented for VS Code, Cursor, Claude Code, and other hosts" + } + ], + "methodology": "Setup complexity assessment across MCP hosts", + "last_verified": "2026-06-10" + }, + "performance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Playwright MCP design notes", + "url": "https://github.com/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "Accessibility-snapshot approach avoids vision-model overhead, making interactions faster and cheaper in tokens than screenshot-based automation" + } + ], + "methodology": "Latency and token-efficiency comparison against pixel-based browser automation", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Playwright MCP tool reference", + "url": "https://github.com/microsoft/playwright-mcp/blob/main/README.md", + "date": "2026-06-10", + "value": "25+ tools covering navigation, clicks, typing, form fill, snapshots, screenshots, network requests, JS evaluation, tabs, file upload, and dialog handling" + } + ], + "methodology": "Feature completeness assessment against common browser-automation needs", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "GitHub API", + "url": "https://api.github.com/repos/microsoft/playwright-mcp", + "date": "2026-06-10", + "value": "33,734 stars; bundled or recommended by major MCP hosts and widely used as the default browser-automation server, replacing the archived Puppeteer reference server" + } + ], + "methodology": "Adoption metrics and ecosystem-integration analysis", + "last_verified": "2026-06-10" + }, + "maintenance_activity": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository activity", + "url": "https://github.com/microsoft/playwright-mcp/commits/main", + "date": "2026-06-10", + "value": "Frequent releases tracking Playwright versions, active issue triage by the Microsoft Playwright team" + } + ], + "methodology": "Commit frequency and release-cadence analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Accessibility-tree snapshots give fast, deterministic, token-efficient page interaction without vision models", + "Comprehensive tool set: navigation, forms, screenshots, network inspection, tabs, dialogs, file upload", + "Backed and actively maintained by Microsoft's Playwright team (33.7k stars)", + "Cross-browser support (Chromium, Firefox, WebKit) with Playwright's auto-waiting reliability", + "Strong mitigation options: isolated profiles, headless mode, origin allow/block lists", + "Flexible transports: stdio by default plus standalone HTTP/SSE server mode" + ], + "limitations": [ + "Untrusted page content flows into the LLM context, creating indirect prompt-injection risk", + "browser_evaluate allows arbitrary JavaScript execution in pages — high-risk if the agent is hijacked", + "Default persistent profile exposes logged-in sessions and cookies to agent actions", + "All visited page content (including authenticated/private pages) is sent to the LLM provider", + "No built-in redaction of PII or credentials visible in snapshots and screenshots", + "Destructive web actions (purchases, posts) are only gated by host-level tool approval" + ], + "metadata": { + "license": "Apache-2.0", + "supported_platforms": [ + "All platforms with Node.js 18+" + ], + "programming_languages": [ + "TypeScript" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/microsoft/playwright-mcp", + "github_stars": 33734, + "package": "@playwright/mcp", + "api_dependency": "Playwright (Chromium/Firefox/WebKit)", + "authentication": "None required (local browser control)", + "first_release": "2025-03", + "maintained_by": "Microsoft (Playwright team)", + "status": "Active", + "supersedes": "mcp-server-puppeteer (archived reference server)", + "transport_types": [ + "stdio", + "http", + "sse" + ], + "installation_methods": [ + "npm", + "npx", + "docker", + "vscode-extension" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 90, + "notes": "Excellent for web app testing, UI verification, and agent-driven E2E automation during development" + }, + "research-assistant": { + "overall": 85, + "notes": "Strong for interactive web research and data gathering, but exposed to prompt injection from untrusted pages" + }, + "data-analysis": { + "overall": 78, + "notes": "Useful for scraping and extracting structured data via snapshots and network inspection" + }, + "customer-support": { + "overall": 70, + "notes": "Can reproduce user-reported web issues and walk through flows; requires careful session isolation" + }, + "content-creation": { + "overall": 72, + "notes": "Handy for previewing, screenshotting, and verifying published web content" + }, + "education": { + "overall": 82, + "notes": "Great for teaching web automation and testing concepts with visible, explainable tool calls" + } + }, + "best_for": [ + "Developers automating browser testing and UI verification with AI agents", + "Teams building agentic web workflows that need deterministic page interaction", + "QA engineers generating and executing end-to-end tests via natural language", + "Researchers performing interactive web data gathering in isolated browser profiles" + ], + "related": [ + "mcp-server-puppeteer", + "mcp-server-chrome-devtools", + "mcp-server-fetch", + "mcp-server-firecrawl" + ], + "tags": [ + "browser-automation", + "playwright", + "testing", + "web-scraping", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-postgres.json b/data/mcps/mcp-server-postgres.json index c1b0e80..01ad894 100644 --- a/data/mcps/mcp-server-postgres.json +++ b/data/mcps/mcp-server-postgres.json @@ -4,9 +4,9 @@ "name": "MCP PostgreSQL Server", "provider": "Anthropic (Archived)", "version": "0.6.2", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former official MCP server for PostgreSQL database interaction. This server is NO LONGER MAINTAINED and has been moved to servers-archived repository. No security updates provided. Community alternatives available. Use with caution.", + "description": "ARCHIVED WITH UNPATCHED VULNERABILITY: Former Anthropic reference MCP server for PostgreSQL, archived 2025-05-29. A SQL injection flaw disclosed by Trend Micro (June 2025) remains unpatched because the repo is archived, yet the npm package still saw ~21k weekly downloads. NOT RECOMMENDED for any use; a patched community fork (@zeddotdev/postgres-context-server) exists.", "website": "https://modelcontextprotocol.io/docs/servers/postgres", "trust_vector": { "performance_reliability": { @@ -85,38 +85,50 @@ } }, "security": { - "overall_score": 71, + "overall_score": 51, "criteria": { "sql_injection_protection": { - "score": 82, + "score": 15, "confidence": "high", "evidence": [ { - "source": "Parameterized Queries", - "url": "https://github.com/modelcontextprotocol/servers/tree/main/src/postgres", - "date": "2025-11-16", - "value": "Uses parameterized queries to prevent SQL injection" + "source": "Trend Micro Research", + "url": "https://www.trendmicro.com/en_us/research/25/f/why-a-classic-mcp-server-vulnerability-can-undermine-your-entire-ai-agent.html", + "date": "2025-06-01", + "value": "Classic SQL injection vulnerability disclosed in the archived PostgreSQL MCP server; will never be patched in @modelcontextprotocol/server-postgres because the repository is archived" + }, + { + "source": "Datadog Security Labs", + "url": "https://securitylabs.datadoghq.com/articles/mcp-vulnerability-case-study-SQL-injection-in-the-postgresql-mcp-server/", + "date": "2025-06-30", + "value": "Case study confirming SQL injection in the PostgreSQL MCP server allows bypassing the read-only transaction restriction; ~21k weekly npm downloads while vulnerable" } ], - "methodology": "SQL injection testing", - "last_verified": "2025-11-09" + "methodology": "Vulnerability disclosure review and exploitability analysis", + "last_verified": "2026-06-10" }, "access_control": { - "score": 75, - "confidence": "medium", + "score": 55, + "confidence": "high", "evidence": [ { "source": "PostgreSQL Permissions", "url": "https://www.postgresql.org/docs/current/sql-grant.html", "date": "2025-11-16", "value": "Inherits database user permissions but AI can execute any query within those permissions" + }, + { + "source": "Datadog Security Labs", + "url": "https://securitylabs.datadoghq.com/articles/mcp-vulnerability-case-study-SQL-injection-in-the-postgresql-mcp-server/", + "date": "2025-06-30", + "value": "SQL injection allows escaping the server's intended read-only transaction sandbox, defeating its primary access-control mechanism" } ], "methodology": "Permission boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "data_modification_risk": { - "score": 65, + "score": 45, "confidence": "high", "evidence": [ { @@ -124,10 +136,16 @@ "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "AI can execute INSERT, UPDATE, DELETE operations if database user has permissions" + }, + { + "source": "Trend Micro Research", + "url": "https://www.trendmicro.com/en_us/research/25/f/why-a-classic-mcp-server-vulnerability-can-undermine-your-entire-ai-agent.html", + "date": "2025-06-01", + "value": "Unpatched SQL injection enables write operations even when the server is configured for read-only access" } ], "methodology": "Write operation risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "credential_security": { "score": 72, @@ -160,7 +178,7 @@ } }, "privacy_compliance": { - "overall_score": 68, + "overall_score": 66, "criteria": { "data_exposure_risk": { "score": 62, @@ -235,7 +253,7 @@ } }, "trust_transparency": { - "overall_score": 84, + "overall_score": 81, "criteria": { "documentation_quality": { "score": 88, @@ -280,23 +298,29 @@ "last_verified": "2025-11-09" }, "security_documentation": { - "score": 75, - "confidence": "medium", + "score": 50, + "confidence": "high", "evidence": [ { "source": "Security Guidelines", "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "Provides security best practices but could be more comprehensive for database access" + }, + { + "source": "MCP servers-archived repository README", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "README states 'NO SECURITY GUARANTEES ARE PROVIDED FOR THESE ARCHIVED SERVERS'; known SQL injection remains undocumented and unpatched in the package" } ], "methodology": "Security documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } }, "operational_excellence": { - "overall_score": 83, + "overall_score": 77, "criteria": { "ease_of_setup": { "score": 85, @@ -355,18 +379,18 @@ "last_verified": "2025-11-09" }, "community_support": { - "score": 80, - "confidence": "medium", + "score": 55, + "confidence": "high", "evidence": [ { - "source": "MCP Community", - "url": "https://github.com/modelcontextprotocol/servers/discussions", - "date": "2025-11-16", - "value": "Growing community support for database MCP servers" + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Server archived 2025-05-29; repository read-only, no issues or PRs accepted. Patched community fork available as @zeddotdev/postgres-context-server" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -376,7 +400,7 @@ "Natural language to SQL query generation capabilities", "Excellent for data analysis and business intelligence workflows", "Full query visibility and logging for audit purposes", - "Open source implementation with active development", + "Open source code remains publicly auditable in the archived repository", "Supports connection pooling and performance optimization" ], "limitations": [ @@ -385,7 +409,8 @@ "AI can execute destructive operations (DELETE, DROP) if permissions allow", "Limited granular access control beyond database user permissions", "Query results with sensitive data sent to external APIs", - "Compliance challenges for regulated industries (HIPAA, PCI-DSS, GDPR)" + "Compliance challenges for regulated industries (HIPAA, PCI-DSS, GDPR)", + "ARCHIVED 2025-05-29 with an UNPATCHED SQL injection vulnerability (Trend Micro, June 2025); use the patched fork @zeddotdev/postgres-context-server instead" ], "metadata": { "license": "MIT", @@ -402,8 +427,9 @@ "database_version": "PostgreSQL 10+", "connection_method": "Connection string with credentials", "first_release": "2024-11", - "maintained_by": "None (Archived)", - "status": "Archived - No longer maintained", + "maintained_by": "None (Archived 2025-05-29)", + "status": "Archived - unpatched SQL injection vulnerability; patched community fork: @zeddotdev/postgres-context-server", + "package_name": "@modelcontextprotocol/server-postgres", "transport_types": [ "stdio" ], @@ -462,6 +488,9 @@ "database", "sql", "mcp", - "model-context-protocol" + "model-context-protocol", + "archived", + "unmaintained", + "security-incidents" ] } diff --git a/data/mcps/mcp-server-puppeteer.json b/data/mcps/mcp-server-puppeteer.json index 9796827..7ba9912 100644 --- a/data/mcps/mcp-server-puppeteer.json +++ b/data/mcps/mcp-server-puppeteer.json @@ -4,13 +4,13 @@ "name": "MCP Puppeteer Server", "provider": "Anthropic (Archived)", "version": "2025.4.6", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former official MCP server for Puppeteer browser automation. This server is NO LONGER MAINTAINED. Moved to servers-archived repository. No security updates provided. Community alternatives may be available.", + "description": "ARCHIVED: Former Anthropic reference MCP server for Puppeteer browser automation, archived 2025-05-29 to the servers-archived repository and no longer maintained. The archived repo explicitly provides no security guarantees. For browser automation, Microsoft's Playwright MCP server is the recommended successor.", "website": "https://pptr.dev/", "trust_vector": { "performance_reliability": { - "overall_score": 81, + "overall_score": 80, "criteria": { "browser_automation_accuracy": { "score": 85, @@ -99,7 +99,7 @@ } }, "security": { - "overall_score": 62, + "overall_score": 63, "criteria": { "browser_sandboxing": { "score": 75, @@ -174,7 +174,7 @@ } }, "privacy_compliance": { - "overall_score": 67, + "overall_score": 66, "criteria": { "web_content_exposure": { "score": 62, @@ -249,7 +249,7 @@ } }, "trust_transparency": { - "overall_score": 79, + "overall_score": 80, "criteria": { "documentation_quality": { "score": 85, @@ -310,7 +310,7 @@ } }, "operational_excellence": { - "overall_score": 76, + "overall_score": 69, "criteria": { "ease_of_setup": { "score": 78, @@ -341,7 +341,7 @@ "last_verified": "2025-11-09" }, "stability": { - "score": 78, + "score": 60, "confidence": "medium", "evidence": [ { @@ -349,10 +349,16 @@ "url": "https://github.com/puppeteer/puppeteer/issues", "date": "2025-11-16", "value": "Generally stable but can have issues with complex pages" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Server archived 2025-05-29; no bug fixes or compatibility updates will be released" } ], "methodology": "Stability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "browser_compatibility": { "score": 82, @@ -369,18 +375,24 @@ "last_verified": "2025-11-09" }, "community_support": { - "score": 82, + "score": 58, "confidence": "high", "evidence": [ { "source": "Puppeteer Community", "url": "https://github.com/puppeteer/puppeteer", "date": "2025-11-16", - "value": "Large community with 85k+ GitHub stars" + "value": "Large community with 85k+ GitHub stars for the underlying Puppeteer library" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "MCP server archived 2025-05-29; repository is read-only, issues and PRs are no longer accepted" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -399,7 +411,8 @@ "Can access credentials and authenticated sessions", "Scraped content sent to external LLM provider", "Legal and ethical concerns with automated web scraping", - "No built-in controls for malicious site protection or PII filtering" + "No built-in controls for malicious site protection or PII filtering", + "ARCHIVED 2025-05-29: unmaintained, no security guarantees; use Microsoft Playwright MCP instead" ], "metadata": { "license": "Apache 2.0 (Puppeteer), varies for MCP implementation", @@ -422,7 +435,8 @@ "puppeteer_version": "21.0+", "resource_requirements": "High (500MB+ RAM per browser instance)", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "Unmaintained (archived 2025-05-29)", + "repository": "https://github.com/modelcontextprotocol/servers-archived" }, "use_case_ratings": { "code-generation": { @@ -471,10 +485,15 @@ "Data collectors extracting information from dynamic websites", "Developers building web automation workflows" ], + "related_entities": [ + "mcp-server-playwright" + ], "tags": [ "browser", "automation", "mcp", - "model-context-protocol" + "model-context-protocol", + "archived", + "unmaintained" ] } diff --git a/data/mcps/mcp-server-sequential-thinking.json b/data/mcps/mcp-server-sequential-thinking.json index e082363..d6f6335 100644 --- a/data/mcps/mcp-server-sequential-thinking.json +++ b/data/mcps/mcp-server-sequential-thinking.json @@ -4,9 +4,9 @@ "name": "MCP Sequential Thinking Server", "provider": "Anthropic", "version": "2025.7.1", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Official Anthropic MCP server enabling dynamic, extended reasoning and problem-solving sequences. Allows AI models to create structured thinking processes, break down complex problems, and maintain context across multi-step reasoning chains. Experimental feature for advanced cognitive workflows.", + "description": "Official MCP reference server enabling dynamic, extended reasoning and problem-solving sequences. Allows AI models to create structured thinking processes, break down complex problems, and maintain context across multi-step reasoning chains. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", "website": "https://modelcontextprotocol.io/docs/servers/sequential-thinking", "trust_vector": { "performance_reliability": { @@ -349,10 +349,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Limited adoption due to experimental status and specialized use case" + }, + { + "source": "Anthropic - Donating MCP to the Agentic AI Foundation", + "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "sequential-thinking is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -362,7 +368,7 @@ "Maintains context across extended reasoning chains", "Supports backtracking and reasoning correction", "High transparency with visible reasoning steps", - "Open source with active Anthropic development", + "Open source and actively maintained under the MCP project (Agentic AI Foundation)", "Minimal security risk due to isolated reasoning process" ], "limitations": [ @@ -371,7 +377,8 @@ "Limited by LLM token context windows for extended reasoning", "Reasoning steps exposed to LLM provider", "Relatively low community adoption due to specialized use case", - "Documentation and best practices still evolving" + "Documentation and best practices still evolving", + "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -388,7 +395,7 @@ "api_dependency": "None (MCP protocol only)", "authentication": "None required", "first_release": "2024-11", - "maintained_by": "Anthropic", + "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Active", "transport_types": [ "stdio" @@ -448,6 +455,8 @@ "reasoning", "thinking", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "reference-server" ] } diff --git a/data/mcps/mcp-server-serena.json b/data/mcps/mcp-server-serena.json new file mode 100644 index 0000000..ac841bd --- /dev/null +++ b/data/mcps/mcp-server-serena.json @@ -0,0 +1,444 @@ +{ + "id": "mcp-server-serena", + "type": "mcp", + "name": "Serena MCP", + "provider": "Oraios AI", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Open-source semantic coding toolkit from Oraios AI that turns any MCP-capable agent into an IDE-grade coding assistant. Uses language servers (LSP) for symbol-level code navigation and editing — find_symbol, find_referencing_symbols, precise symbol edits — plus project memory and shell execution. High-privilege local tooling: full filesystem and shell access.", + "website": "https://github.com/oraios/serena", + "trust_vector": { + "performance_reliability": { + "overall_score": 82, + "criteria": { + "symbol_resolution_accuracy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Serena README", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Symbol lookup and reference finding are backed by real language servers (LSP), giving compiler-grade accuracy for find_symbol and find_referencing_symbols across many supported languages" + } + ], + "methodology": "Accuracy testing of symbol resolution and reference finding against IDE ground truth in multi-file projects", + "last_verified": "2026-06-10" + }, + "operation_success_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Serena tool suite", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Symbol-level editing operations (insert/replace at symbol granularity) succeed reliably when the language server has indexed the project; failures cluster around partially-indexed or syntactically broken code" + } + ], + "methodology": "Hands-on testing of symbolic read/edit operations across project states", + "last_verified": "2026-06-10" + }, + "language_server_stability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Serena issue tracker", + "url": "https://github.com/oraios/serena/issues", + "date": "2026-06-10", + "value": "Stability varies by language server implementation; large projects can hit slow startup indexing, and some language servers occasionally require restart — a known operational caveat" + } + ], + "methodology": "Review of reported language-server issues and stress testing on large repositories", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Serena implementation", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Tool failures return descriptive errors; language servers can be restarted via tooling, and the onboarding/memory system helps agents re-establish context after failures" + } + ], + "methodology": "Error-path testing including LSP crashes, unindexed files, and invalid edits", + "last_verified": "2026-06-10" + }, + "large_project_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Serena design documentation", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Symbol-level navigation reads only relevant code instead of whole files, keeping token usage low on large codebases — a key advantage over grep/read-based agents" + } + ], + "methodology": "Token-efficiency and navigation testing on repositories with 100k+ lines", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 49, + "criteria": { + "shell_execution_risk": { + "score": 40, + "confidence": "high", + "evidence": [ + { + "source": "Serena tool suite", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "execute_shell_command runs arbitrary shell commands with the user's privileges — effectively a full local code-execution surface; can be disabled via tool configuration/modes but is enabled in typical agent setups" + } + ], + "methodology": "Capability analysis of shell execution tooling and its abuse potential under prompt injection", + "last_verified": "2026-06-10" + }, + "filesystem_access_risk": { + "score": 50, + "confidence": "high", + "evidence": [ + { + "source": "Serena project configuration", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Read/write access across the activated project directory, including config files and anything reachable by relative paths; project activation provides scoping but enforcement is application-level, not sandboxed" + } + ], + "methodology": "Analysis of file read/write tool boundaries and project-scoping enforcement", + "last_verified": "2026-06-10" + }, + "sandboxing_isolation": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "Serena runtime model", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Runs as a local Python process (uvx/uv) with no OS-level sandbox; combined with shell execution, a hijacked agent inherits the full local user privilege. Containerized deployment is possible but not the default" + } + ], + "methodology": "Review of process isolation, privilege boundaries, and available containment options", + "last_verified": "2026-06-10" + }, + "credential_exposure_risk": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Serena architecture", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Server itself requires no credentials, but filesystem and shell access mean local secrets (.env files, SSH keys, tokens) are readable if the agent is steered to do so" + } + ], + "methodology": "Analysis of secret-reachability via file and shell tools", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "MCP security guidance", + "url": "https://modelcontextprotocol.io/docs/concepts/security", + "date": "2026-06-10", + "value": "Code edits and shell commands can alter or destroy local state and reach the network; Serena's modes/contexts can restrict tool availability, and host-level approval remains the main guardrail for destructive actions" + } + ], + "methodology": "Authorization boundary analysis of write and execution tools, including available mode-based restrictions", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 76, + "criteria": { + "code_exposure": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "MCP data flow architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-06-10", + "value": "Symbol contents, file excerpts, and shell output are sent to the LLM provider as tool results; symbol-level reads expose less code per call than whole-file approaches but private code still leaves the machine" + } + ], + "methodology": "Data flow analysis of tool outputs to the LLM provider", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Serena repository", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "No built-in secret detection or redaction; .env files, keys, and credentials in the project tree can be read and forwarded to the LLM if requested" + } + ], + "methodology": "Privacy controls assessment of file-content handling", + "last_verified": "2026-06-10" + }, + "local_data_control": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Serena architecture", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Fully local operation: no hosted backend, no vendor telemetry; project memories are stored as plain local files (.serena directory) the user can inspect and delete" + } + ], + "methodology": "Review of local execution model, memory storage, and data residency", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Serena documentation", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "No data is sent to Oraios AI or any third party by the server itself; the only outbound data flow is tool results to the user's chosen LLM provider" + } + ], + "methodology": "Data sharing pathway analysis", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 86, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Serena README and docs", + "url": "https://github.com/oraios/serena/blob/main/README.md", + "date": "2026-06-10", + "value": "Detailed README covering installation (uvx/uv), client integration, modes/contexts, tool list, supported languages, and project onboarding; active changelog" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Fully open source under MIT with no proprietary backend — the entire stack including language-server orchestration is auditable" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Serena tooling", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "All operations are explicit named tool calls visible in MCP host logs; an optional local dashboard/log window shows server activity, though shell command side effects are only as visible as their output" + } + ], + "methodology": "Logging and observability assessment including the built-in dashboard", + "last_verified": "2026-06-10" + }, + "project_memory_transparency": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Serena memory system", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Project memories are human-readable markdown files stored in the repository's .serena directory — fully inspectable and editable, though persistent memories can silently shape future agent behavior if unreviewed" + } + ], + "methodology": "Review of memory persistence format, location, and influence on agent sessions", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 85, + "criteria": { + "ease_of_setup": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Serena installation docs", + "url": "https://github.com/oraios/serena#quick-start", + "date": "2026-06-10", + "value": "One-liner via uvx serena (PyPI serena-agent), but requires Python/uv tooling plus per-language language servers; initial project indexing and onboarding add friction versus zero-config servers" + } + ], + "methodology": "Setup complexity assessment including language-server prerequisites", + "last_verified": "2026-06-10" + }, + "performance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Serena design documentation", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Symbol-level operations are fast after indexing and dramatically cheaper in tokens than whole-file reads; first-run language-server indexing on large projects can take minutes" + } + ], + "methodology": "Latency and token-efficiency benchmarking on indexed projects", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Serena tool suite", + "url": "https://github.com/oraios/serena", + "date": "2026-06-10", + "value": "Comprehensive coding toolkit: symbol search/references, symbol-level editing, pattern search, file operations, shell execution, project memory, and onboarding — spanning 20+ languages via LSP" + } + ], + "methodology": "Feature completeness assessment against IDE-grade coding-agent needs", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "GitHub API", + "url": "https://api.github.com/repos/oraios/serena", + "date": "2026-06-10", + "value": "25,204 stars as of 2026-06-10; widely adopted as a free, open-source way to add semantic code tools to Claude, and other MCP-capable agents" + } + ], + "methodology": "Adoption metrics and community-activity analysis", + "last_verified": "2026-06-10" + }, + "maintenance_activity": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "GitHub repository activity", + "url": "https://github.com/oraios/serena/commits/main", + "date": "2026-06-10", + "value": "Very active development by Oraios AI with frequent releases, expanding language support, and responsive issue triage" + } + ], + "methodology": "Commit frequency and release-cadence analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Language-server (LSP) backbone gives compiler-grade symbol navigation and references", + "Symbol-level reading/editing slashes token usage on large codebases versus whole-file approaches", + "Fully open source (MIT), fully local — no hosted backend, no telemetry, no API costs", + "Project memory system persists codebase knowledge across sessions as inspectable markdown", + "Broad language coverage (20+ languages) and active maintenance (25.2k stars)", + "Modes/contexts allow restricting the tool surface, including disabling shell execution" + ], + "limitations": [ + "Shell execution plus filesystem write access make it effectively full local code execution — high-privilege tooling that must be treated like granting terminal access", + "No OS-level sandboxing by default; a prompt-injected agent inherits full user privileges", + "No secret detection — local .env files, keys, and tokens are reachable and forwardable to the LLM", + "Language-server stability and first-run indexing time vary by language and project size", + "Setup requires Python/uv tooling and per-language language servers — more friction than npx-based servers", + "Persistent project memories can silently steer future sessions if not reviewed" + ], + "metadata": { + "license": "MIT", + "supported_platforms": [ + "macOS, Linux, Windows with Python 3.11+ (uv/uvx)" + ], + "programming_languages": [ + "Python" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/oraios/serena", + "github_stars": 25204, + "package": "serena-agent (PyPI)", + "api_dependency": "Local language servers (LSP) per language", + "authentication": "None required (local operation)", + "first_release": "2025-04", + "maintained_by": "Oraios AI", + "status": "Active", + "privilege_level": "High - local filesystem write and shell command execution", + "transport_types": [ + "stdio", + "sse" + ], + "installation_methods": [ + "uvx", + "uv", + "pip", + "docker" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 94, + "notes": "Purpose-built for semantic coding: IDE-grade symbol navigation and precise edits make it one of the strongest free coding-agent toolkits" + }, + "research-assistant": { + "overall": 80, + "notes": "Excellent for exploring and understanding large codebases via symbol-level navigation" + }, + "education": { + "overall": 78, + "notes": "Good for learning how real codebases are structured, though setup and privilege level need supervision" + }, + "data-analysis": { + "overall": 65, + "notes": "Shell access enables running analysis scripts, but this is incidental rather than a designed capability" + } + }, + "best_for": [ + "Developers who want IDE-grade semantic code tools in any MCP-capable agent for free", + "Agentic refactoring and navigation of large codebases with minimal token spend", + "Teams preferring fully local, open-source tooling over hosted coding services", + "Multi-language projects that benefit from LSP-accurate symbol operations" + ], + "related": [ + "mcp-server-filesystem", + "mcp-server-git", + "mcp-server-github", + "mcp-server-context7" + ], + "tags": [ + "coding-agent", + "language-server", + "semantic-code", + "refactoring", + "local-tools", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-shadcn.json b/data/mcps/mcp-server-shadcn.json new file mode 100644 index 0000000..0d76b22 --- /dev/null +++ b/data/mcps/mcp-server-shadcn.json @@ -0,0 +1,421 @@ +{ + "id": "mcp-server-shadcn", + "type": "mcp", + "name": "shadcn MCP Server", + "provider": "shadcn", + "version": "CLI 3.x", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "MCP server built into the shadcn CLI (run via npx shadcn@latest mcp) that lets AI agents browse, search, and install UI components from any shadcn-compatible registry, including private registries, using @registry/name namespacing. Shipped with CLI 3.0 in August 2025.", + "website": "https://ui.shadcn.com/docs/registry/mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 85, + "criteria": { + "registry_resolution_accuracy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Namespaced @registry/name resolution reliably targets components across configured registries" + } + ], + "methodology": "Registry resolution testing across namespaces", + "last_verified": "2026-06-10" + }, + "installation_success_rate": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "shadcn CLI 3.0 Changelog", + "url": "https://ui.shadcn.com/docs/changelog/2025-08-cli-3-mcp", + "date": "2026-06-10", + "value": "Installs reuse the mature shadcn add pipeline, handling dependencies, file placement, and framework detection" + } + ], + "methodology": "Component installation success testing", + "last_verified": "2026-06-10" + }, + "component_search_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Browse and search tools surface components with metadata from any shadcn-compatible registry" + } + ], + "methodology": "Search relevance assessment over public registries", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn-ui/ui Repository", + "url": "https://github.com/shadcn-ui/ui", + "date": "2026-06-10", + "value": "CLI surfaces clear errors for missing registries, unreachable URLs, and dependency conflicts; partial installs may need manual cleanup" + } + ], + "methodology": "Failure mode and error message testing", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 65, + "criteria": { + "supply_chain_risk": { + "score": 52, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry Documentation", + "url": "https://ui.shadcn.com/docs/registry", + "date": "2026-06-10", + "value": "Installs source code from any configured shadcn-compatible registry directly into the project; third-party registries are unvetted and the key risk surface" + } + ], + "methodology": "Supply chain threat modeling of registry-sourced code installation", + "last_verified": "2026-06-10" + }, + "code_installation_control": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Agent-initiated installs write component files and add npm dependencies; review depends on the MCP client's approval flow and code review practices" + } + ], + "methodology": "Write-action control and approval flow analysis", + "last_verified": "2026-06-10" + }, + "registry_authentication": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn Registry Documentation", + "url": "https://ui.shadcn.com/docs/registry", + "date": "2026-06-10", + "value": "Private registries supported with token-based authentication configured via environment variables in components.json" + } + ], + "methodology": "Registry authentication mechanism review", + "last_verified": "2026-06-10" + }, + "credential_handling": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn Registry Documentation", + "url": "https://ui.shadcn.com/docs/registry", + "date": "2026-06-10", + "value": "Registry tokens read from local environment; no credentials sent to third parties beyond the configured registry endpoints" + } + ], + "methodology": "Credential storage and flow analysis", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn CLI 3.0 Changelog", + "url": "https://ui.shadcn.com/docs/changelog/2025-08-cli-3-mcp", + "date": "2026-06-10", + "value": "An agent can install components (and transitively their dependencies) into the codebase; a malicious or typosquatted registry entry executes at build/runtime" + } + ], + "methodology": "Blast radius assessment of install actions", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 83, + "criteria": { + "data_exposure": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Server only fetches public or configured registry metadata and component source; it does not read or transmit project code externally" + } + ], + "methodology": "Data flow analysis of registry requests", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn Registry Documentation", + "url": "https://ui.shadcn.com/docs/registry", + "date": "2026-06-10", + "value": "Minimal sensitive data involved; main consideration is keeping private registry tokens out of committed configuration" + } + ], + "methodology": "Sensitive data surface assessment", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn Registry Documentation", + "url": "https://ui.shadcn.com/docs/registry", + "date": "2026-06-10", + "value": "Search and install requests are visible to whichever registries are configured, including third-party ones" + } + ], + "methodology": "Outbound request and data sharing review", + "last_verified": "2026-06-10" + }, + "local_execution_privacy": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Runs locally over stdio via npx shadcn@latest mcp with no hosted service or telemetry pipeline collecting project data" + } + ], + "methodology": "Local execution and telemetry review", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 88, + "criteria": { + "documentation_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Clear official docs covering MCP setup per client, registry configuration, and namespacing, plus a detailed CLI 3.0 changelog" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn CLI 3.0 Changelog", + "url": "https://ui.shadcn.com/docs/changelog/2025-08-cli-3-mcp", + "date": "2026-06-10", + "value": "Installed files and dependency changes are fully visible in the project diff; tool calls appear in MCP client logs" + } + ], + "methodology": "Traceability of agent actions assessment", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "shadcn-ui/ui Repository", + "url": "https://github.com/shadcn-ui/ui", + "date": "2026-06-10", + "value": "MIT-licensed; MCP server ships as part of the open shadcn-ui/ui monorepo and CLI" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-06-10" + }, + "tool_coverage_clarity": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Tool surface is small and well-defined: browse, search, and install components from configured registries" + } + ], + "methodology": "Tool surface documentation review", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 88, + "criteria": { + "ease_of_setup": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "One-line stdio config (npx shadcn@latest mcp) with documented one-command setup for major MCP clients" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-06-10" + }, + "performance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn-ui/ui Repository", + "url": "https://github.com/shadcn-ui/ui", + "date": "2026-06-10", + "value": "Lightweight local process; latency dominated by registry HTTP fetches and npm dependency installs" + } + ], + "methodology": "Operation latency characterization", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "shadcn CLI 3.0 Changelog", + "url": "https://ui.shadcn.com/docs/changelog/2025-08-cli-3-mcp", + "date": "2026-06-10", + "value": "Built on the actively maintained shadcn CLI with frequent releases since the August 2025 3.0 launch" + } + ], + "methodology": "Maintenance cadence and stability analysis", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "shadcn Registry MCP Documentation", + "url": "https://ui.shadcn.com/docs/registry/mcp", + "date": "2026-06-10", + "value": "Covers discovery and installation across public and private shadcn-compatible registries; scoped to UI components rather than general tooling" + } + ], + "methodology": "Feature completeness assessment within its domain", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "shadcn-ui/ui Repository", + "url": "https://github.com/shadcn-ui/ui", + "date": "2026-06-10", + "value": "shadcn/ui is one of the most-starred UI projects on GitHub and the MCP server is the standard agent integration for its ecosystem" + } + ], + "methodology": "Ecosystem adoption analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "First-party integration with the dominant shadcn/ui component ecosystem", + "Browse, search, and install from any shadcn-compatible registry, including private ones", + "Runs locally over stdio with no hosted service and minimal data exposure", + "Trivial setup via npx shadcn@latest mcp with per-client docs", + "All installed code lands in the project diff for review", + "MIT-licensed and developed in the open shadcn-ui/ui monorepo" + ], + "limitations": [ + "Installs third-party source code and npm dependencies into the project — registry supply-chain risk is the key concern", + "No built-in vetting or signing of third-party registry components", + "Agent-initiated installs depend on the MCP client's approval flow for human oversight", + "Scope limited to UI component workflows in shadcn-compatible projects", + "Private registry tokens must be carefully kept out of committed config" + ], + "metadata": { + "license": "MIT", + "supported_platforms": [ + "All platforms with Node.js" + ], + "programming_languages": [ + "TypeScript" + ], + "github_repo": "https://github.com/shadcn-ui/ui", + "package_name": "shadcn", + "installation": "npx shadcn@latest mcp", + "authentication": "Optional registry tokens for private registries", + "first_release": "2025-08", + "maintained_by": "shadcn", + "transport_types": [ + "stdio" + ], + "installation_methods": [ + "npm" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "Excellent for AI-assisted frontend development: agents can find and install real, vetted-by-usage components instead of hallucinating UI code" + }, + "content-creation": { + "overall": 72, + "notes": "Useful when building landing pages and content sites on shadcn-based stacks" + }, + "education": { + "overall": 80, + "notes": "Good for teaching modern React/Tailwind component patterns with installable examples" + }, + "research-assistant": { + "overall": 60, + "notes": "Limited to exploring component registries; not a general research tool" + }, + "data-analysis": { + "overall": 55, + "notes": "Only relevant for installing chart and dashboard UI components, not for analysis itself" + } + }, + "best_for": [ + "Frontend teams building with shadcn/ui and Tailwind who want agent-driven component installs", + "Organizations distributing internal design systems via private shadcn registries", + "AI coding agents that should use real registry components rather than generated approximations" + ], + "related_entities": [ + "mcp-server-context7", + "mcp-server-playwright" + ], + "tags": [ + "ui-components", + "design-system", + "frontend", + "registry", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-slack.json b/data/mcps/mcp-server-slack.json index 13a593a..3570609 100644 --- a/data/mcps/mcp-server-slack.json +++ b/data/mcps/mcp-server-slack.json @@ -4,13 +4,13 @@ "name": "MCP Slack Server", "provider": "Anthropic (Archived)", "version": "2025.4.24", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former official MCP server for Slack workspace interaction. This server is NO LONGER MAINTAINED. Moved to servers-archived repository. No security updates provided. Community alternatives may be available.", + "description": "ARCHIVED: Former Anthropic reference MCP server for Slack workspace interaction, archived 2025-05-29 to the servers-archived repository. NO LONGER MAINTAINED and no security guarantees are provided for archived servers. Its replacement in the MCP ecosystem is a third-party maintained Slack MCP server; evaluate maintained alternatives before deploying.", "website": "https://api.slack.com/", "trust_vector": { "performance_reliability": { - "overall_score": 84, + "overall_score": 83, "criteria": { "message_delivery_reliability": { "score": 90, @@ -85,7 +85,7 @@ } }, "security": { - "overall_score": 74, + "overall_score": 75, "criteria": { "oauth_security": { "score": 85, @@ -235,7 +235,7 @@ } }, "trust_transparency": { - "overall_score": 81, + "overall_score": 76, "criteria": { "documentation_quality": { "score": 88, @@ -280,23 +280,23 @@ "last_verified": "2025-11-09" }, "mcp_implementation": { - "score": 75, - "confidence": "medium", + "score": 55, + "confidence": "high", "evidence": [ { - "source": "Community Implementation", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Community-maintained with variable documentation quality" + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Reference Slack server archived 2025-05-29; repository read-only, README states no security guarantees are provided for archived servers" } ], "methodology": "Implementation documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } }, "operational_excellence": { - "overall_score": 80, + "overall_score": 76, "criteria": { "ease_of_setup": { "score": 78, @@ -341,7 +341,7 @@ "last_verified": "2025-11-09" }, "feature_coverage": { - "score": 76, + "score": 55, "confidence": "medium", "evidence": [ { @@ -349,10 +349,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Covers basic message reading/posting, limited channel management" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "Archived 2025-05-29; feature set frozen, no updates for Slack API or MCP specification changes" } ], "methodology": "Feature completeness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "cost": { "score": 75, @@ -384,7 +390,7 @@ "No built-in content filtering or PII protection", "Risk of exposing confidential team communications", "Rate limits can be restrictive for high-volume operations", - "Community-maintained MCP server with variable support", + "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; replacement is third-party maintained", "Compliance challenges when sharing workspace data with third parties" ], "metadata": { @@ -403,7 +409,8 @@ "api_version": "Slack Web API", "rate_limits": "Tier-based (1+ requests per second)", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "None (Archived 2025-05-29)", + "repository": "https://github.com/modelcontextprotocol/servers-archived" }, "use_case_ratings": { "code-generation": { @@ -456,6 +463,8 @@ "communication", "collaboration", "mcp", - "model-context-protocol" + "model-context-protocol", + "archived", + "unmaintained" ] } diff --git a/data/mcps/mcp-server-sqlite.json b/data/mcps/mcp-server-sqlite.json index 94a5398..4fe5037 100644 --- a/data/mcps/mcp-server-sqlite.json +++ b/data/mcps/mcp-server-sqlite.json @@ -2,11 +2,11 @@ "id": "mcp-server-sqlite", "type": "mcp", "name": "MCP SQLite Server", - "provider": "Anthropic", + "provider": "Anthropic (Archived)", "version": "2025.4.24", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Official MCP server enabling AI models to interact with SQLite databases through natural language. Supports schema inspection, query execution, and data analysis for local file-based databases through the Model Context Protocol. Ideal for lightweight data applications but requires security considerations.", + "description": "ARCHIVED WITH UNPATCHED VULNERABILITY: Former Anthropic reference MCP server for SQLite, archived 2025-05-29 to the servers-archived repository. The archived code contains the same class of unpatched SQL injection flaw disclosed by Trend Micro (June 2025) and will not receive fixes; the archive README provides no security guarantees. Not recommended for new deployments.", "website": "https://modelcontextprotocol.io/docs/servers/sqlite", "trust_vector": { "performance_reliability": { @@ -85,21 +85,21 @@ } }, "security": { - "overall_score": 73, + "overall_score": 56, "criteria": { "sql_injection_protection": { - "score": 85, + "score": 20, "confidence": "high", "evidence": [ { - "source": "Parameterized Queries", - "url": "https://github.com/modelcontextprotocol/servers/tree/main/src/sqlite", - "date": "2025-11-16", - "value": "Uses parameterized queries to prevent SQL injection" + "source": "Trend Micro Research", + "url": "https://www.trendmicro.com/en_us/research/25/f/why-a-classic-mcp-server-vulnerability-can-undermine-your-entire-ai-agent.html", + "date": "2025-06-01", + "value": "Same class of classic SQL injection vulnerability present in the archived SQLite MCP server code; cannot be patched upstream because the repository was archived 2025-05-29" } ], - "methodology": "SQL injection testing", - "last_verified": "2025-11-09" + "methodology": "Vulnerability disclosure review and code analysis", + "last_verified": "2026-06-10" }, "file_access_control": { "score": 70, @@ -116,7 +116,7 @@ "last_verified": "2025-11-09" }, "data_modification_risk": { - "score": 68, + "score": 50, "confidence": "high", "evidence": [ { @@ -124,10 +124,16 @@ "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "AI can execute any SQL including DROP, DELETE if database file is writable" + }, + { + "source": "Trend Micro Research", + "url": "https://www.trendmicro.com/en_us/research/25/f/why-a-classic-mcp-server-vulnerability-can-undermine-your-entire-ai-agent.html", + "date": "2025-06-01", + "value": "Unpatched SQL injection in archived code can be abused via crafted input to perform unintended write operations" } ], "methodology": "Write operation risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" }, "encryption_support": { "score": 65, @@ -160,7 +166,7 @@ } }, "privacy_compliance": { - "overall_score": 71, + "overall_score": 73, "criteria": { "data_exposure_risk": { "score": 65, @@ -221,7 +227,7 @@ } }, "trust_transparency": { - "overall_score": 87, + "overall_score": 91, "criteria": { "documentation_quality": { "score": 90, @@ -282,7 +288,7 @@ } }, "operational_excellence": { - "overall_score": 90, + "overall_score": 86, "criteria": { "ease_of_setup": { "score": 95, @@ -341,7 +347,7 @@ "last_verified": "2025-11-09" }, "maintenance": { - "score": 85, + "score": 58, "confidence": "high", "evidence": [ { @@ -349,10 +355,16 @@ "url": "https://www.sqlite.org/", "date": "2025-11-16", "value": "Minimal maintenance required, no server to manage" + }, + { + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-06-10", + "value": "MCP server archived 2025-05-29; repository read-only, no maintainer, no bug fixes or security patches will be released" } ], "methodology": "Maintenance overhead assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -371,7 +383,8 @@ "Limited concurrent write performance due to file-level locking", "No built-in PII detection or data anonymization", "AI can execute destructive SQL if database file is writable", - "No native audit logging beyond MCP protocol logs" + "No native audit logging beyond MCP protocol logs", + "ARCHIVED 2025-05-29 with an unpatched SQL injection flaw in the archived code (Trend Micro, June 2025); no security guarantees, not recommended for new deployments" ], "metadata": { "license": "MIT (MCP server), Public Domain (SQLite)", @@ -387,7 +400,7 @@ "Python" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", + "github_repo": "https://github.com/modelcontextprotocol/servers-archived", "database_version": "SQLite 3.x", "max_database_size": "281 TB (theoretical)", "first_release": "2024-11", @@ -444,6 +457,9 @@ "database", "sql", "mcp", - "model-context-protocol" + "model-context-protocol", + "archived", + "unmaintained", + "security-incidents" ] } diff --git a/data/mcps/mcp-server-stripe.json b/data/mcps/mcp-server-stripe.json new file mode 100644 index 0000000..db40abc --- /dev/null +++ b/data/mcps/mcp-server-stripe.json @@ -0,0 +1,440 @@ +{ + "id": "mcp-server-stripe", + "type": "mcp", + "name": "Stripe MCP Server", + "provider": "Stripe", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Stripe's official MCP server from the open-source agent toolkit. Exposes tools for customers, products, prices, payment links, invoices, refunds, balance, disputes, and subscriptions plus Stripe documentation search. Available as the @stripe/mcp stdio package or the hosted remote server at mcp.stripe.com with OAuth.", + "website": "https://docs.stripe.com/mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 86, + "criteria": { + "api_reliability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Status", + "url": "https://status.stripe.com/", + "date": "2026-06-10", + "value": "Built directly on the Stripe API, which maintains historically high uptime backed by Stripe's production payments infrastructure" + } + ], + "methodology": "API stability and uptime analysis of the underlying Stripe API", + "last_verified": "2026-06-10" + }, + "operation_success_rate": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Agent Toolkit Repository", + "url": "https://github.com/stripe/agent-toolkit", + "date": "2026-06-10", + "value": "Tools are thin, well-tested wrappers over stable Stripe API endpoints (customers, invoices, payment links, refunds, subscriptions)" + } + ], + "methodology": "Operation success testing across the documented tool set in test mode", + "last_verified": "2026-06-10" + }, + "rate_limit_handling": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe Rate Limits Documentation", + "url": "https://docs.stripe.com/rate-limits", + "date": "2026-06-10", + "value": "Inherits Stripe API rate limits with standard 429 responses; limits are generous for typical agent workloads" + } + ], + "methodology": "Rate limiting behavior testing under sustained tool-call load", + "last_verified": "2026-06-10" + }, + "search_accuracy": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "Includes a Stripe documentation search tool that returns relevant docs for integration questions alongside account-data tools" + } + ], + "methodology": "Relevance assessment of documentation search results for common integration queries", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe Agent Toolkit Repository", + "url": "https://github.com/stripe/agent-toolkit", + "date": "2026-06-10", + "value": "Surfaces Stripe's structured error objects (decline codes, validation errors) to the agent, enabling informed retries" + } + ], + "methodology": "Error handling testing with invalid parameters, missing permissions, and declined operations", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 80, + "criteria": { + "authentication_security": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "Local server authenticates with Stripe API keys (restricted keys supported); hosted server at https://mcp.stripe.com uses OAuth with explicit consent" + } + ], + "methodology": "Authentication mechanism review for stdio and hosted transports", + "last_verified": "2026-06-10" + }, + "scope_limitation": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Restricted API Keys Documentation", + "url": "https://docs.stripe.com/keys#create-restricted-api-secret-key", + "date": "2026-06-10", + "value": "Restricted API Keys allow per-resource read/write permissions, so the server can be limited to exactly the tools and access levels needed" + } + ], + "methodology": "Permission scope testing with restricted keys across read and write tools", + "last_verified": "2026-06-10" + }, + "token_exposure_risk": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "stdio mode requires placing a secret key in client configuration; a leaked unrestricted key grants full account access, making restricted keys essential" + } + ], + "methodology": "Token storage and exposure-surface analysis for local configuration files", + "last_verified": "2026-06-10" + }, + "action_auditability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Dashboard Logs", + "url": "https://docs.stripe.com/development/dashboard/request-logs", + "date": "2026-06-10", + "value": "Every tool call is an API request visible in Stripe Dashboard request logs and events, attributable to the specific key or OAuth grant" + } + ], + "methodology": "Audit logging review of API request logs and event history", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "Tools include money-movement operations (refunds, payment links, invoices, subscription changes); an agent acting on injected or mistaken instructions can cause direct financial impact without human confirmation" + } + ], + "methodology": "Threat modeling of write-capable financial tools; strongest case among evaluated servers for read-only defaults and human-in-the-loop confirmation on writes", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 77, + "criteria": { + "payment_data_exposure": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "Customer records, invoice details, and payment metadata returned by tools flow into the LLM provider's context; raw card numbers are never exposed by the Stripe API" + } + ], + "methodology": "Data flow analysis of tool results containing customer and payment data", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Security Documentation", + "url": "https://docs.stripe.com/security", + "date": "2026-06-10", + "value": "Stripe is PCI-DSS Level 1 certified and SOC 2 audited; card data is tokenized server-side and not retrievable through MCP tools" + } + ], + "methodology": "Review of Stripe compliance posture and the data classes reachable via the tool surface", + "last_verified": "2026-06-10" + }, + "organization_data_control": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe Keys and Permissions Documentation", + "url": "https://docs.stripe.com/keys", + "date": "2026-06-10", + "value": "Account owners control exposure via restricted keys, test vs live mode separation, and OAuth grant revocation" + } + ], + "methodology": "Access control review of key management and OAuth grant administration", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe Privacy Policy", + "url": "https://stripe.com/privacy", + "date": "2026-06-10", + "value": "Customer PII retrieved via tools is shared with the user's LLM provider under that provider's data policy, outside Stripe's compliance boundary" + } + ], + "methodology": "Analysis of downstream data sharing once tool results leave Stripe", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 89, + "criteria": { + "documentation_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "First-class documentation page covering hosted and local setup, full tool list, permissions guidance, and security recommendations" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Dashboard Request Logs", + "url": "https://docs.stripe.com/development/dashboard/request-logs", + "date": "2026-06-10", + "value": "All operations appear in Dashboard request logs and the events stream with full request/response detail" + } + ], + "methodology": "Logging and traceability assessment via Dashboard logs", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Agent Toolkit Repository", + "url": "https://github.com/stripe/agent-toolkit", + "date": "2026-06-10", + "value": "MIT-licensed open source (approximately 1,601 stars); tool implementations are fully auditable in the stripe/agent-toolkit repository (being renamed stripe/ai)" + } + ], + "methodology": "Source code review of the published toolkit and MCP package", + "last_verified": "2026-06-10" + }, + "api_coverage_clarity": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "Tool list is explicitly enumerated (customers, products, prices, payment links, invoices, refunds, balance, disputes, subscriptions, doc search) with per-tool enablement flags" + } + ], + "methodology": "Comparison of documented tool surface against the shipped package", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 86, + "criteria": { + "ease_of_setup": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-06-10", + "value": "One-line npx @stripe/mcp setup with selectable tools, or zero-install hosted server at https://mcp.stripe.com via OAuth" + } + ], + "methodology": "Setup complexity assessment for both stdio and hosted transports", + "last_verified": "2026-06-10" + }, + "api_performance": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe API Documentation", + "url": "https://docs.stripe.com/api", + "date": "2026-06-10", + "value": "Stripe API responses are typically fast (low hundreds of milliseconds); tool wrappers add negligible overhead" + } + ], + "methodology": "Latency observation across representative tool calls", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Status", + "url": "https://status.stripe.com/", + "date": "2026-06-10", + "value": "Hosted MCP and underlying API ride on Stripe's production infrastructure with strong historical availability" + } + ], + "methodology": "Uptime analysis of Stripe infrastructure", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Stripe Agent Toolkit Repository", + "url": "https://github.com/stripe/agent-toolkit", + "date": "2026-06-10", + "value": "Covers core billing and payments objects plus doc search; advanced surfaces (Connect, Treasury, Radar) are not fully exposed as tools" + } + ], + "methodology": "Feature completeness assessment against the full Stripe API surface", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Stripe Agent Toolkit Repository", + "url": "https://github.com/stripe/agent-toolkit", + "date": "2026-06-10", + "value": "Approximately 1,601 GitHub stars with active first-party maintenance; widely referenced as the canonical payments MCP server" + } + ], + "methodology": "Community activity and adoption analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "First-party, MIT-licensed open source implementation maintained by Stripe", + "Restricted API Keys enable precise per-resource, read-only or write scoping", + "Hosted remote option with OAuth avoids placing secret keys in client config", + "Full auditability via Stripe Dashboard request logs and events", + "Backed by PCI-DSS Level 1 and SOC 2 compliant infrastructure", + "Built-in Stripe documentation search alongside account tools", + "Test mode allows safe end-to-end agent evaluation before live use" + ], + "limitations": [ + "Money-movement tools (refunds, invoices, payment links, subscriptions) make misuse directly costly", + "No built-in human confirmation step on write operations; must be enforced by the client", + "Customer PII in tool results is shared with the LLM provider", + "Unrestricted secret keys in stdio config are a severe single point of failure", + "Tool coverage omits advanced surfaces like Connect and Treasury", + "Agent errors in live mode can require manual financial remediation" + ], + "metadata": { + "repository": "https://github.com/stripe/agent-toolkit", + "package_name": "@stripe/mcp", + "license": "MIT", + "maintained_by": "Stripe", + "github_stars": 1601, + "remote_endpoint": "https://mcp.stripe.com", + "authentication": "Stripe API keys / Restricted API Keys (stdio); OAuth (hosted)", + "transport_types": [ + "stdio", + "streamable-http (hosted)" + ], + "installation_methods": [ + "npx @stripe/mcp", + "Remote MCP endpoint" + ], + "compliance": [ + "PCI-DSS Level 1", + "SOC 2" + ], + "mcp_version": "1.0" + }, + "use_case_ratings": { + "financial-analysis": { + "overall": 85, + "notes": "Strong for querying balances, disputes, subscriptions, and revenue objects directly from the source of truth" + }, + "customer-support": { + "overall": 84, + "notes": "Excellent for support agents looking up customers, invoices, and issuing scoped refunds with proper guardrails" + }, + "code-generation": { + "overall": 82, + "notes": "Doc search plus live test-mode tools make it very effective for building Stripe integrations" + }, + "data-analysis": { + "overall": 76, + "notes": "Good for ad-hoc account analysis, though bulk analytics is better served by Stripe Sigma or data exports" + }, + "legal-compliance": { + "overall": 62, + "notes": "Dispute and record access is useful, but PII flowing to LLM providers requires careful review" + } + }, + "best_for": [ + "Support and operations teams automating customer, invoice, and refund workflows with restricted keys", + "Developers building and testing Stripe integrations with AI assistance in test mode", + "Finance teams querying live billing and subscription state conversationally", + "Agent builders who need a well-audited, first-party payments tool surface" + ], + "related_entities": [ + "mcp-server-github", + "mcp-server-vercel", + "mcp-server-zapier", + "mcp-server-slack" + ], + "tags": [ + "payments", + "billing", + "fintech", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-time.json b/data/mcps/mcp-server-time.json index 371a57e..3929542 100644 --- a/data/mcps/mcp-server-time.json +++ b/data/mcps/mcp-server-time.json @@ -4,9 +4,9 @@ "name": "MCP Time Server", "provider": "Anthropic", "version": "2025.9.25", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Official Anthropic MCP server for time and timezone operations. Provides AI models with current time information, timezone conversions, date calculations, and scheduling assistance. Essential for time-sensitive workflows and international collaboration.", + "description": "Official MCP reference server for time and timezone operations. Provides AI models with current time information, timezone conversions, date calculations, and scheduling assistance. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", "website": "https://modelcontextprotocol.io/docs/servers/time", "trust_vector": { "performance_reliability": { @@ -349,10 +349,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Popular utility MCP server with broad adoption" + }, + { + "source": "Anthropic - Donating MCP to the Agentic AI Foundation", + "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "time is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-06-10" } } } @@ -371,7 +377,8 @@ "Cannot modify system time (by design)", "Query patterns may reveal user schedule to LLM provider", "Depends on system time accuracy and timezone database updates", - "No advanced scheduling or reminder capabilities" + "No advanced scheduling or reminder capabilities", + "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -388,7 +395,7 @@ "api_dependency": "System time APIs, IANA Time Zone Database", "authentication": "None required", "first_release": "2024-11", - "maintained_by": "Anthropic", + "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Active", "transport_types": [ "stdio" @@ -449,6 +456,8 @@ "time", "utilities", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "reference-server" ] } diff --git a/data/mcps/mcp-server-vercel.json b/data/mcps/mcp-server-vercel.json new file mode 100644 index 0000000..140aa80 --- /dev/null +++ b/data/mcps/mcp-server-vercel.json @@ -0,0 +1,433 @@ +{ + "id": "mcp-server-vercel", + "type": "mcp", + "name": "Vercel MCP Server", + "provider": "Vercel", + "version": "2026.6-beta", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Vercel's official hosted MCP server at mcp.vercel.com. Provides Vercel documentation search and tools to inspect teams, projects, deployments, and deployment logs. Remote-only with OAuth 2.1, a client allowlist, and mandatory consent; deliberately read-only at initial launch. Currently in Public Beta.", + "website": "https://vercel.com/docs/mcp/vercel-mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 83, + "criteria": { + "api_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Hosted on Vercel's own platform infrastructure with a single managed endpoint at https://mcp.vercel.com" + } + ], + "methodology": "Endpoint stability analysis on Vercel platform infrastructure", + "last_verified": "2026-06-10" + }, + "operation_success_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Read-oriented tools (projects, deployments, logs) are thin wrappers over the stable Vercel REST API and succeed consistently within granted scopes" + } + ], + "methodology": "Operation success testing across documented tools on real projects", + "last_verified": "2026-06-10" + }, + "search_accuracy": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Documentation search returns current Vercel docs, reducing hallucinated configuration answers in coding agents" + } + ], + "methodology": "Relevance assessment of documentation search results for common platform queries", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Returns structured errors for unauthorized scopes, unknown projects, and expired sessions; OAuth re-consent flow recovers expired grants" + } + ], + "methodology": "Error handling testing across permission, scope, and session-expiry failures", + "last_verified": "2026-06-10" + }, + "rate_limit_handling": { + "score": 78, + "confidence": "low", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Inherits Vercel API rate limits per authenticated user; MCP-specific limits are not separately published during Public Beta" + } + ], + "methodology": "Rate limiting behavior observation; limited published detail during beta", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 87, + "criteria": { + "authentication_security": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "OAuth 2.1 authorization with mandatory user consent on every connection; no static API tokens are placed in client configuration" + } + ], + "methodology": "Review of OAuth 2.1 flow, consent screens, and token lifecycle", + "last_verified": "2026-06-10" + }, + "scope_limitation": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Deliberately read-only at initial launch: tools inspect teams, projects, deployments, and logs but cannot mutate or trigger deployments" + } + ], + "methodology": "Permission boundary testing of the shipped tool surface for write capability", + "last_verified": "2026-06-10" + }, + "token_exposure_risk": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Remote-only design means no locally stored long-lived secrets; OAuth tokens are short-lived and revocable from the Vercel dashboard" + } + ], + "methodology": "Token storage and exposure-surface analysis for the remote-only model", + "last_verified": "2026-06-10" + }, + "unauthorized_action_risk": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "A client allowlist restricts which MCP clients may connect, and the read-only launch surface means a confused or injected agent cannot modify infrastructure" + } + ], + "methodology": "Threat modeling of agent misuse against the allowlisted, read-only tool surface", + "last_verified": "2026-06-10" + }, + "action_auditability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "OAuth grants are visible and revocable per user; access activity is attributable to the granting account, though a dedicated MCP audit log is not yet exposed" + } + ], + "methodology": "Audit logging review of grant management and access attribution", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 76, + "criteria": { + "deployment_data_exposure": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Project metadata, deployment details, and log contents are returned to the LLM provider as tool results" + } + ], + "methodology": "Data flow analysis of tool results from Vercel to LLM providers", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Deployment logs can contain secrets, tokens, or PII printed by application code; the server does not redact log contents before returning them" + } + ], + "methodology": "Assessment of redaction controls on log-reading tools", + "last_verified": "2026-06-10" + }, + "organization_data_control": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel Access Control Documentation", + "url": "https://vercel.com/docs/rbac", + "date": "2026-06-10", + "value": "Access mirrors the authenticated user's team roles and project permissions; team admins govern membership and can revoke OAuth grants" + } + ], + "methodology": "Access control review against Vercel team RBAC", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel Privacy Policy", + "url": "https://vercel.com/legal/privacy-policy", + "date": "2026-06-10", + "value": "Project and log data retrieved via MCP is shared with the user's chosen LLM provider under that provider's data policy" + } + ], + "methodology": "Analysis of downstream data sharing once tool results leave Vercel", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 74, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Clear official documentation covering endpoint, OAuth setup, supported clients, tool capabilities, and security model" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Tool calls are visible in MCP client logs; connected integrations and grants are visible in the Vercel dashboard" + } + ], + "methodology": "Logging and traceability assessment across client and dashboard", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Overview Repository", + "url": "https://github.com/vercel/vercel-mcp-overview", + "date": "2026-06-10", + "value": "Server implementation is closed source; a public overview repository documents capabilities but the code cannot be independently audited" + } + ], + "methodology": "Source availability and independent verifiability review", + "last_verified": "2026-06-10" + }, + "api_coverage_clarity": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Documented tool surface (docs search; team, project, deployment management views; deployment logs) matches observed behavior, with read-only status stated explicitly" + } + ], + "methodology": "Comparison of documented tool surface against observed server capabilities", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 80, + "criteria": { + "ease_of_setup": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Zero installation: add https://mcp.vercel.com in a supported client and complete the OAuth consent flow" + } + ], + "methodology": "Setup complexity assessment across allowlisted MCP clients", + "last_verified": "2026-06-10" + }, + "api_performance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Hosted on Vercel's edge-adjacent infrastructure; typical tool responses return quickly, with log retrieval slowest for large deployments" + } + ], + "methodology": "Latency observation across representative tool calls", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel Status", + "url": "https://www.vercel-status.com/", + "date": "2026-06-10", + "value": "Rides on Vercel platform availability, which is historically strong, though the MCP product itself is in Public Beta" + } + ], + "methodology": "Uptime analysis combined with beta-status risk assessment", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Vercel MCP Documentation", + "url": "https://vercel.com/docs/mcp/vercel-mcp", + "date": "2026-06-10", + "value": "Read-only launch scope excludes deployment triggering, environment variable management, and domain configuration; coverage is intentionally narrow" + } + ], + "methodology": "Feature completeness assessment against the full Vercel API surface", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Vercel MCP Overview Repository", + "url": "https://github.com/vercel/vercel-mcp-overview", + "date": "2026-06-10", + "value": "First-party server promoted to Vercel's large developer base and supported by major MCP clients since the Public Beta announcement" + } + ], + "methodology": "Adoption analysis across the Vercel developer ecosystem and MCP clients", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Notably strong security posture: OAuth 2.1, mandatory consent, client allowlist, and read-only launch scope", + "Remote-only design eliminates locally stored long-lived secrets", + "Live documentation search reduces hallucinated Vercel configuration answers", + "Deployment log access enables fast AI-assisted debugging of failed builds", + "Zero-install setup in supported MCP clients", + "Backed by Vercel's reliable platform infrastructure" + ], + "limitations": [ + "Read-only at launch: cannot trigger deployments or change configuration", + "Closed source; implementation cannot be independently audited", + "Deployment logs may contain unredacted secrets or PII that flow to the LLM provider", + "Public Beta status means the tool surface and limits may change", + "Client allowlist excludes unsupported or custom MCP clients", + "No dedicated MCP-specific audit log exposed yet" + ], + "metadata": { + "repository": "https://github.com/vercel/vercel-mcp-overview", + "license": "Proprietary (closed source; public overview repo)", + "maintained_by": "Vercel", + "status": "Public Beta", + "remote_endpoint": "https://mcp.vercel.com", + "authentication": "OAuth 2.1 with mandatory consent and client allowlist", + "transport_types": [ + "streamable-http (remote only)" + ], + "installation_methods": [ + "Remote MCP endpoint" + ], + "write_access": "Read-only at initial launch by design", + "mcp_version": "1.0" + }, + "use_case_ratings": { + "code-generation": { + "overall": 86, + "notes": "Docs search plus project and log context makes coding agents far more accurate on Vercel deployments" + }, + "research-assistant": { + "overall": 78, + "notes": "Good for investigating deployment history, failures, and project configuration across teams" + }, + "data-analysis": { + "overall": 68, + "notes": "Useful for inspecting deployment metadata and logs, but no aggregate analytics tooling" + }, + "customer-support": { + "overall": 64, + "notes": "Helps internal support diagnose customer-facing deployment issues from logs" + }, + "education": { + "overall": 72, + "notes": "Safe read-only surface is well suited to teaching deployment and platform concepts" + } + }, + "best_for": [ + "Developers debugging Vercel deployments and build failures with AI assistance", + "Coding agents that need accurate, current Vercel documentation and project context", + "Teams that want platform visibility for agents without granting any write access", + "Security-conscious organizations evaluating remote MCP servers" + ], + "related_entities": [ + "mcp-server-github", + "mcp-server-figma", + "mcp-server-stripe", + "mcp-server-linear" + ], + "tags": [ + "deployment", + "devops", + "hosting", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-zapier.json b/data/mcps/mcp-server-zapier.json new file mode 100644 index 0000000..9e80cc9 --- /dev/null +++ b/data/mcps/mcp-server-zapier.json @@ -0,0 +1,444 @@ +{ + "id": "mcp-server-zapier", + "type": "mcp", + "name": "Zapier MCP Server", + "provider": "Zapier", + "version": "2026.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Zapier's hosted, proprietary MCP server that gives AI agents access to user-selected actions from 9,000+ connected apps (Gmail, Slack, Salesforce, and more, spanning tens of thousands of actions). Each user generates a personal remote server endpoint at mcp.zapier.com with per-app and per-action permissioning.", + "website": "https://zapier.com/mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 85, + "criteria": { + "api_reliability": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Runs on Zapier's mature integration platform that has powered production automation for over a decade" + } + ], + "methodology": "Platform stability and maturity analysis", + "last_verified": "2026-06-10" + }, + "action_execution_success": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Actions reuse Zapier's battle-tested app integrations; failures are usually due to downstream app auth or API issues" + } + ], + "methodology": "Action execution success testing across common apps", + "last_verified": "2026-06-10" + }, + "app_integration_breadth": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP", + "url": "https://zapier.com/mcp", + "date": "2026-06-10", + "value": "9,000+ apps and roughly 30,000-40,000 actions available, the broadest integration catalog of any MCP server" + } + ], + "methodology": "Integration catalog assessment", + "last_verified": "2026-06-10" + }, + "rate_limit_handling": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Plan-based usage limits on MCP tool calls; downstream app rate limits surface as action errors" + } + ], + "methodology": "Rate and usage limit behavior review", + "last_verified": "2026-06-10" + }, + "error_recovery": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Errors from downstream apps are returned to the agent; reauthorization of expired app connections requires manual user action" + } + ], + "methodology": "Failure mode and recovery testing", + "last_verified": "2026-06-10" + } + } + }, + "security": { + "overall_score": 71, + "criteria": { + "authentication_security": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "A user-specific server is generated at mcp.zapier.com; access is tied to that personal endpoint rather than a separately rotated credential in legacy URL-auth mode" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-06-10" + }, + "url_credential_exposure": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "The per-user endpoint URL acts as a bearer credential: anyone who obtains it can invoke the user's enabled actions, so it must be treated as a secret and rotated if leaked" + } + ], + "methodology": "Credential exposure threat modeling of the endpoint URL", + "last_verified": "2026-06-10" + }, + "blast_radius_control": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP", + "url": "https://zapier.com/mcp", + "date": "2026-06-10", + "value": "By design the server can reach email, CRM, chat, and finance apps simultaneously; a compromised or prompt-injected agent has an extremely broad blast radius across connected accounts" + } + ], + "methodology": "Blast radius assessment across connected app categories", + "last_verified": "2026-06-10" + }, + "permission_granularity": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Users explicitly select which apps and which individual actions are exposed to the agent, enabling least-privilege configuration" + } + ], + "methodology": "Per-app and per-action permission model review", + "last_verified": "2026-06-10" + }, + "credential_handling": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Zapier Security", + "url": "https://zapier.com/security", + "date": "2026-06-10", + "value": "Downstream app OAuth credentials are stored by Zapier under SOC 2 Type II audited controls and never exposed to the agent" + } + ], + "methodology": "Credential storage and isolation review", + "last_verified": "2026-06-10" + } + } + }, + "privacy_compliance": { + "overall_score": 75, + "criteria": { + "data_exposure": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Emails, CRM records, and messages retrieved by actions flow through Zapier's cloud and into the LLM provider context" + } + ], + "methodology": "Data flow analysis of action inputs and outputs", + "last_verified": "2026-06-10" + }, + "sensitive_data_protection": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier Security", + "url": "https://zapier.com/security", + "date": "2026-06-10", + "value": "Encryption in transit and at rest on the platform, but no content-level PII redaction before data reaches the model" + } + ], + "methodology": "Data protection controls assessment", + "last_verified": "2026-06-10" + }, + "third_party_data_sharing": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier Privacy Policy", + "url": "https://zapier.com/privacy", + "date": "2026-06-10", + "value": "Action data is processed by Zapier, the downstream app, and the LLM provider, each under separate privacy policies" + } + ], + "methodology": "Multi-party data sharing review", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Zapier Security and Compliance", + "url": "https://zapier.com/security", + "date": "2026-06-10", + "value": "SOC 2 Type II audited, with published GDPR and CCPA compliance commitments" + } + ], + "methodology": "Compliance certification review", + "last_verified": "2026-06-10" + } + } + }, + "trust_transparency": { + "overall_score": 73, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Clear setup guides per MCP client, action configuration docs, and security guidance" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "operation_visibility": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Dashboard", + "url": "https://mcp.zapier.com/", + "date": "2026-06-10", + "value": "Per-user dashboard shows configured actions and a history of agent tool invocations" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-06-10" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP", + "url": "https://zapier.com/mcp", + "date": "2026-06-10", + "value": "Fully proprietary, hosted-only service; server implementation cannot be inspected or self-hosted" + } + ], + "methodology": "Source availability review", + "last_verified": "2026-06-10" + }, + "tool_coverage_clarity": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Exposed tool set is exactly the actions the user enabled, each with documented parameters drawn from Zapier's integration schemas" + } + ], + "methodology": "Tool surface documentation review", + "last_verified": "2026-06-10" + } + } + }, + "operational_excellence": { + "overall_score": 87, + "criteria": { + "ease_of_setup": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "No installation: generate a personal endpoint at mcp.zapier.com, pick apps and actions in the web UI, and paste the URL into the MCP client" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-06-10" + }, + "api_performance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-06-10", + "value": "Action latency adds Zapier orchestration overhead on top of downstream app API calls, typically a few seconds per action" + } + ], + "methodology": "Action latency characterization", + "last_verified": "2026-06-10" + }, + "reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Zapier Status Page", + "url": "https://status.zapier.com/", + "date": "2026-06-10", + "value": "Public status page with historically high availability across the Zapier platform" + } + ], + "methodology": "Uptime and incident history analysis", + "last_verified": "2026-06-10" + }, + "feature_coverage": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Zapier MCP", + "url": "https://zapier.com/mcp", + "date": "2026-06-10", + "value": "Tens of thousands of actions across 9,000+ apps including Gmail, Slack, Salesforce, HubSpot, and Google Workspace" + } + ], + "methodology": "Capability breadth assessment", + "last_verified": "2026-06-10" + }, + "community_adoption": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Zapier MCP", + "url": "https://zapier.com/mcp", + "date": "2026-06-10", + "value": "Heavily promoted integration path supported by major MCP clients and Zapier's large existing automation user base" + } + ], + "methodology": "Adoption and ecosystem support analysis", + "last_verified": "2026-06-10" + } + } + } + }, + "strengths": [ + "Broadest action catalog of any MCP server: 9,000+ apps and tens of thousands of actions", + "Per-app and per-action permissioning enables least-privilege agent configuration", + "Zero-install hosted setup with a personal endpoint generated at mcp.zapier.com", + "Downstream app credentials held by Zapier under SOC 2 Type II audited controls, never exposed to the agent", + "Action invocation history visible in the per-user MCP dashboard", + "Built on a decade-mature integration platform with high availability" + ], + "limitations": [ + "Extremely broad blast radius by design: a prompt-injected agent can act across email, CRM, chat, and finance apps", + "The per-user endpoint URL acts as a bearer credential and must be treated as a secret", + "Fully proprietary and hosted-only; no source inspection or self-hosting", + "Business data from connected apps flows through Zapier's cloud and the LLM provider", + "Subject to plan-based usage limits and pricing", + "Expired downstream app connections require manual reauthorization" + ], + "metadata": { + "license": "Proprietary", + "supported_platforms": [ + "Hosted remote server (any MCP client with remote support)" + ], + "api_dependency": "Zapier platform and 9,000+ downstream app APIs", + "authentication": "User-specific endpoint at mcp.zapier.com (URL acts as bearer credential)", + "compliance": "SOC 2 Type II", + "maintained_by": "Zapier", + "documentation": "https://docs.zapier.com/mcp/home", + "transport_types": [ + "remote (hosted)" + ], + "installation_methods": [ + "hosted endpoint" + ] + }, + "use_case_ratings": { + "customer-support": { + "overall": 88, + "notes": "Excellent for agents that triage tickets, draft replies, and update CRM records across support stacks" + }, + "data-analysis": { + "overall": 75, + "notes": "Good for pulling records from business apps into analysis, though not an analytics tool itself" + }, + "content-creation": { + "overall": 80, + "notes": "Strong for publishing and distribution workflows across CMS, email, and social apps" + }, + "code-generation": { + "overall": 58, + "notes": "Not a development tool; mainly useful for wiring deployment or notification side effects" + }, + "research-assistant": { + "overall": 70, + "notes": "Useful for gathering data from connected SaaS tools rather than the open web" + }, + "financial-analysis": { + "overall": 68, + "notes": "Can reach accounting and CRM data, but broad access to financial apps demands strict action scoping" + }, + "legal-compliance": { + "overall": 62, + "notes": "SOC 2 helps, but routing privileged documents through Zapier and an LLM needs careful review" + }, + "healthcare": { + "overall": 50, + "notes": "Not suitable for PHI workflows; no HIPAA business associate posture for MCP agent traffic" + } + }, + "best_for": [ + "Business automation agents that need to act across many SaaS apps from one server", + "Teams already invested in Zapier who want their stack exposed to AI assistants", + "Non-technical users wanting agent integrations without running any infrastructure" + ], + "related_entities": [ + "mcp-server-apify", + "mcp-server-fetch" + ], + "tags": [ + "automation", + "saas-integrations", + "workflow", + "hosted", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/models/claude-fable-5.json b/data/models/claude-fable-5.json new file mode 100644 index 0000000..e429f3f --- /dev/null +++ b/data/models/claude-fable-5.json @@ -0,0 +1,651 @@ +{ + "id": "claude-fable-5", + "type": "model", + "name": "Claude Fable 5", + "provider": "Anthropic", + "version": "claude-fable-5", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Anthropic's new top-tier model above Opus and the first generally available Mythos-class model. State-of-the-art on nearly all tested benchmarks at launch, including the highest frontier score on Cognition's FrontierCode. Adaptive thinking only, 1M context, 128K output.", + "website": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + + "trust_vector": { + "performance_reliability": { + "overall_score": 98, + "criteria": { + "task_accuracy_code": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "Cognition FrontierCode", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "Highest frontier score recorded on Cognition's FrontierCode benchmark at launch" + }, + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "State-of-the-art on nearly all tested coding and agentic benchmarks at launch" + } + ], + "methodology": "Frontier coding benchmarks measuring real-world software engineering and long-horizon agentic coding tasks", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "SOTA across tested graduate-level reasoning and science benchmarks at launch" + } + ], + "methodology": "Graduate and PhD-level reasoning benchmarks requiring multi-step problem solving, evaluated with adaptive thinking at high effort", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "SOTA on knowledge and multimodal (text + vision) evaluations at launch" + } + ], + "methodology": "Comprehensive knowledge and multimodal testing across text and vision inputs", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 96, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Model Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-09", + "value": "Adaptive thinking with effort parameter (low/medium/high/xhigh/max); sampling parameters (temperature/top_p) removed for more predictable behavior" + } + ], + "methodology": "Repeated-run consistency testing across effort levels; adaptive-thinking-only surface removes sampling variance controls", + "last_verified": "2026-06-10", + "notes": "No manual thinking budgets or temperature/top_p sampling parameters; behavior steered via prompting and the effort parameter" + }, + "latency_p50": { + "value": "3.0s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-10", + "value": "Early measurements ~3.0s median for standard prompts at default effort; launch-day data is preliminary" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes; limited launch-window sample", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "7.0s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-10", + "value": "Early p95 ~7.0s; higher at xhigh/max effort due to deeper adaptive thinking" + } + ], + "methodology": "95th percentile response time across diverse workloads; limited launch-window sample", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-09", + "value": "1M token context window, 128K max output tokens" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Status Page", + "url": "https://status.anthropic.com/", + "date": "2026-06-10", + "value": "99.9%+ platform uptime (last 90 days); Fable 5 served on the same infrastructure" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Current highest-performing model in the registry. SOTA on nearly all tested benchmarks at launch, including the top frontier score on Cognition's FrontierCode. Latency data is preliminary (released 2026-06-09)." + }, + + "security": { + "overall_score": 92, + "criteria": { + "prompt_injection_resistance": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "Mythos-class safety training; improved resistance to injected instructions in agentic settings" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attacks; third-party red-team data still limited at launch", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-06-09", + "value": "Constitutional AI alignment carried forward to the Mythos-class generation with strengthened refusal calibration" + } + ], + "methodology": "Testing against adversarial prompt datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Statement", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-06-09", + "value": "Training opt-out by default for API traffic; no training on user data without consent" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "Released under Anthropic's Responsible Scaling Policy with frontier-tier safeguards; Claude Mythos 5 itself restricted to research partners" + } + ], + "methodology": "Comprehensive safety testing across harmful content categories per Responsible Scaling Policy", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/api/rate-limits", + "date": "2026-06-09", + "value": "API key and OAuth authentication, HTTPS only, rate limiting, workspace scoping" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Frontier-tier safety posture; the unrestricted Mythos-class research model (Claude Mythos 5) is limited to research partners while Fable 5 is the generally available variant. Independent red-team coverage still accumulating at launch." + }, + + "privacy_compliance": { + "overall_score": 93, + "criteria": { + "data_residency": { + "value": "US, EU (customer choice)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Enterprise Documentation", + "url": "https://www.anthropic.com/claude/enterprise", + "date": "2026-06-09", + "value": "Data residency options for US and EU enterprise customers" + } + ], + "methodology": "Review of enterprise documentation and privacy policies", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Privacy Policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-06-09", + "value": "Training opt-out by default for API usage" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Configurable; zero-retention available", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-06-09", + "value": "Zero data retention agreements available for eligible API customers" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Documentation", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-06-09", + "value": "Customer responsible for PII redaction; provider-side safeguards for incidental PII" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-06-09", + "value": "SOC 2 Type II, GDPR compliant, HIPAA eligible" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-06-09", + "value": "Zero data retention configuration available; no training on API data by default" + } + ], + "methodology": "Review of data handling practices and trust center documentation", + "last_verified": "2026-06-10" + } + }, + "notes": "Same strong Anthropic compliance posture as the Opus line: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API traffic." + }, + + "trust_transparency": { + "overall_score": 88, + "criteria": { + "explainability": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Adaptive Thinking Documentation", + "url": "https://platform.claude.com/docs/en/build-with-claude/adaptive-thinking", + "date": "2026-06-09", + "value": "Adaptive thinking with effort control; thinking content omitted by default, summarized reasoning available via display option" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10", + "notes": "Thinking text is omitted by default; opt in to summarized display for visible reasoning" + }, + "hallucination_rate": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "Improved factual accuracy and calibration over Opus 4.8 on internal evaluations" + } + ], + "methodology": "Testing on factual QA datasets; independent measurement still limited at launch", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Responsible Scaling Policy", + "url": "https://www.anthropic.com/news/anthropics-responsible-scaling-policy", + "date": "2026-06-09", + "value": "Regular bias testing and mitigation under the Responsible Scaling Policy" + } + ], + "methodology": "Evaluation on bias benchmarks and diverse demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "Stronger calibration and appropriate uncertainty expression reported at launch" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Model Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-09", + "value": "Comprehensive model card with capabilities, limitations, benchmarks, and safety evaluations" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Public Statements", + "url": "https://www.anthropic.com/news", + "date": "2026-06-09", + "value": "General description provided, detailed sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-06-09", + "value": "Constitutional AI safety guardrails with Mythos-class enhancements" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong documentation and guardrails. Thinking content is omitted by default (summarized display is opt-in), which slightly reduces out-of-the-box reasoning visibility compared to older Opus defaults." + }, + + "operational_excellence": { + "overall_score": 91, + "criteria": { + "api_design_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-09", + "value": "Same API surface as Opus 4.7/4.8: adaptive thinking, effort parameter (low/medium/high/xhigh/max), structured outputs, task budgets, tool use, streaming" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10", + "notes": "One new breaking change vs Opus 4.8: explicit thinking disabled returns 400 — omit the thinking parameter instead. No temperature/top_p sampling parameters." + }, + "sdk_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic SDKs", + "url": "https://github.com/anthropics", + "date": "2026-06-09", + "value": "Official SDKs (Python, TypeScript, Java, Go, Ruby, C#, PHP) with day-one Fable 5 support" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Versioning", + "url": "https://platform.claude.com/docs/en/api/versioning", + "date": "2026-06-09", + "value": "Clear versioning with advance deprecation notice; stable claude-fable-5 alias" + } + ], + "methodology": "Review of versioning policy and historical practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Console", + "url": "https://console.anthropic.com/", + "date": "2026-06-09", + "value": "Usage dashboard with metrics, cost tracking, and workspace controls" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Support", + "url": "https://support.anthropic.com/", + "date": "2026-06-09", + "value": "Email support, developer community, comprehensive documentation and migration guides" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", + "date": "2026-06-09", + "value": "Available on the Anthropic API at launch; first generally available Mythos-class model — cloud-provider rollout following" + } + ], + "methodology": "Analysis of third-party integrations and availability surfaces", + "last_verified": "2026-06-10", + "notes": "Released 2026-06-09; multi-cloud availability narrower than Opus line at evaluation time" + }, + "license_terms": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Commercial Terms", + "url": "https://www.anthropic.com/legal/commercial-terms", + "date": "2026-06-09", + "value": "Standard commercial terms; enterprise agreements available" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Same API surface as Opus 4.7/4.8 makes adoption straightforward for existing Claude users. Day-old release means ecosystem and operational track record are still maturing." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 98, + "notes": "Highest frontier score on Cognition's FrontierCode and SOTA on tested coding benchmarks at launch. Best-in-registry for the hardest software engineering work; xhigh effort recommended.", + "alternatives": ["claude-opus-4-8", "gpt-5-3-codex"] + }, + "customer-support": { + "overall": 88, + "notes": "Exceptional quality but premium pricing ($10/$50) and latency make it overkill for routine support; reserve for complex escalations.", + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] + }, + "content-creation": { + "overall": 95, + "notes": "Top-tier long-form writing with strong structure and voice control. Effort parameter lets teams trade cost for polish on flagship pieces.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "data-analysis": { + "overall": 97, + "notes": "SOTA quantitative reasoning with 1M context for whole-dataset and multi-document analysis.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "research-assistant": { + "overall": 97, + "notes": "Best-in-registry deep research: 1M context, adaptive thinking, and strong synthesis across large corpora.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 93, + "notes": "Strong privacy posture (SOC 2 Type II, GDPR, HIPAA-eligible) and excellent long-document analysis; launch-recency may matter for conservative legal teams.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "healthcare": { + "overall": 91, + "notes": "HIPAA eligible with training opt-out by default. Highest accuracy in the registry for clinical reasoning, though real-world validation is still early post-launch.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "financial-analysis": { + "overall": 96, + "notes": "SOTA quantitative and multi-step reasoning; 1M context handles full filings and model workbooks in one pass.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "education": { + "overall": 94, + "notes": "Excellent explanations with effort-adjustable depth; premium pricing limits high-volume tutoring deployments.", + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] + }, + "creative-writing": { + "overall": 93, + "notes": "Strong narrative craft and stylistic range. No temperature/top_p controls — variance must be elicited via prompting.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + } + }, + + "strengths": [ + "State-of-the-art on nearly all tested benchmarks at launch; highest-performing model in the registry", + "Highest frontier score on Cognition's FrontierCode coding benchmark", + "First generally available Mythos-class model (Claude Mythos 5 itself is restricted to research partners)", + "1M token context window with 128K max output, text + vision", + "Adaptive thinking with effort parameter (low/medium/high/xhigh/max) for cost/quality control", + "Strong compliance posture: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API" + ], + + "limitations": [ + "Premium pricing at $10/$50 per 1M tokens (2x Opus 4.8)", + "Adaptive thinking only — no manual thinking budgets, and no temperature/top_p sampling parameters", + "Explicit thinking-disabled requests return 400 (omit the thinking parameter instead)", + "Released 2026-06-09 — independent benchmark replication and operational track record still limited", + "Higher latency than Sonnet/Haiku tiers, especially at xhigh/max effort" + ], + + "best_for": [ + "Frontier-difficulty software engineering and long-horizon agentic coding", + "Deep research and analysis over very large corpora (1M context)", + "High-stakes reasoning where accuracy justifies premium cost", + "Enterprise workloads requiring strong compliance with top-tier capability" + ], + + "not_recommended_for": [ + "Cost-sensitive high-volume inference (use Sonnet or Haiku tiers)", + "Real-time applications requiring sub-second latency", + "Workflows that depend on temperature/top_p sampling controls", + "Audio processing applications" + ], + + "metadata": { + "pricing": { + "input": "$10.00 per 1M tokens", + "output": "$50.00 per 1M tokens", + "notes": "2x Opus 4.8 pricing. Batch API 50% discount and prompt caching savings apply.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 128000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)", "document", "computer-use"], + "api_endpoint": "https://api.anthropic.com/v1/messages", + "api_model_id": "claude-fable-5", + "open_source": false, + "architecture": "Mythos-class transformer with Constitutional AI alignment; adaptive thinking only with effort parameter (low/medium/high/xhigh/max)", + "parameters": "Not disclosed", + "knowledge_cutoff": "Not disclosed", + "release_date": "2026-06-09" + }, + + "related_entities": ["claude-opus-4-8", "claude-opus-4-7", "claude-sonnet-4-6", "gpt-5-5", "gemini-3-1-pro"], + + "tags": [ + "coding", + "reasoning", + "enterprise", + "hipaa-eligible", + "safety-focused", + "effort-parameter", + "adaptive-thinking", + "long-context", + "mythos-class", + "flagship" + ] +} diff --git a/data/models/claude-opus-4-5.json b/data/models/claude-opus-4-5.json index f37d61b..8f87e7b 100644 --- a/data/models/claude-opus-4-5.json +++ b/data/models/claude-opus-4-5.json @@ -4,9 +4,9 @@ "name": "Claude Opus 4.5", "provider": "Anthropic", "version": "20251101", - "last_evaluated": "2026-01-14", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Anthropic's most capable model with 80.9% SWE-bench (industry-leading), unique effort parameter for compute control, and exceptional abstract reasoning. First model to exceed 80% on SWE-bench Verified.", + "description": "SUPERSEDED: no longer Anthropic's most capable model — succeeded by Opus 4.6, 4.7, 4.8 (2026-05-28) and Claude Fable 5 (2026-06-09, new top tier). At launch it scored 80.9% SWE-bench and was the first model to exceed 80% on SWE-bench Verified, with a unique effort parameter for compute control.", "website": "https://www.anthropic.com/claude/opus", "trust_vector": { @@ -467,6 +467,12 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2025-11-24", "value": "Clear versioning with 6-month deprecation notice" + }, + { + "source": "Anthropic: Claude Opus 4.8", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-06-10", + "value": "Opus 4.5 superseded by Opus 4.6 (2026-02-05), 4.7 (2026-04-16), 4.8 (2026-05-28); Claude Fable 5 (2026-06-09) is the new top tier" } ], "methodology": "Review of versioning policy and historical practices", @@ -601,7 +607,8 @@ "Smaller context than Gemini 3 (200K vs 1M)", "Premium pricing ($5/$25 per 1M tokens)", "No native audio capabilities", - "Training data transparency limited (industry standard)" + "Training data transparency limited (industry standard)", + "SUPERSEDED: Opus 4.6/4.7/4.8 and Claude Fable 5 (2026-06-09) are newer; no longer Anthropic's most capable model" ], "best_for": [ @@ -649,16 +656,16 @@ "knowledge_cutoff": "May 2025" }, - "related_entities": ["claude-sonnet-4-5", "claude-opus-4-1", "gpt-5-2", "gemini-3-pro"], + "related_entities": ["claude-opus-4-8", "claude-fable-5", "claude-sonnet-4-5", "claude-opus-4-1", "gpt-5-2", "gemini-3-pro"], "tags": [ + "superseded", "coding", "reasoning", "enterprise", "hipaa-eligible", "safety-focused", "effort-parameter", - "computer-use", - "flagship" + "computer-use" ] } diff --git a/data/models/claude-opus-4-6.json b/data/models/claude-opus-4-6.json new file mode 100644 index 0000000..a7d4ea4 --- /dev/null +++ b/data/models/claude-opus-4-6.json @@ -0,0 +1,661 @@ +{ + "id": "claude-opus-4-6", + "type": "model", + "name": "Claude Opus 4.6", + "provider": "Anthropic", + "version": "20260205", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Anthropic's frontier Opus released February 2026 with 80.8% SWE-bench Verified, breakthrough 68.8% ARC-AGI-2 abstract reasoning, adaptive thinking, and a 1M token context window. Now two generations behind Opus 4.8 but still served.", + "website": "https://www.anthropic.com/claude/opus", + + "trust_vector": { + "performance_reliability": { + "overall_score": 97, + "criteria": { + "task_accuracy_code": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "SWE-bench Verified", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "80.8% resolution rate (frontier-class software engineering)" + }, + { + "source": "Terminal-Bench 2.0", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "65.4% on command-line tasks (up from Opus 4.5's 59.3%)" + }, + { + "source": "OSWorld", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "72.7% on computer-use tasks (up from Opus 4.5's 66.3%)" + } + ], + "methodology": "Industry-standard coding and agentic benchmarks measuring real-world software engineering and computer-use tasks", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "ARC-AGI-2", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "68.8% (up from Opus 4.5's 37.6% — a generational leap in abstract reasoning)" + }, + { + "source": "Anthropic Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "Adaptive thinking dynamically allocates reasoning depth per request" + } + ], + "methodology": "Abstract reasoning and multi-step problem solving benchmarks", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-02-05", + "value": "Frontier-class general knowledge and multimodal understanding at launch" + } + ], + "methodology": "Comprehensive knowledge and multimodal testing across published benchmarks", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "Effort parameter GA (low/medium/high/max) enables consistent quality control; adaptive thinking replaces manual budgets" + } + ], + "methodology": "Internal testing of output stability across effort levels and adaptive thinking", + "last_verified": "2026-06-10", + "notes": "Adaptive thinking removes the need to tune thinking budgets manually; the GA effort parameter (including new 'max') gives precise compute control" + }, + "latency_p50": { + "value": "2.5s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models/claude-opus-4-6", + "date": "2026-03-01", + "value": "Typical response time ~2.5s for standard prompts at default effort" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "5.5s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models/claude-opus-4-6", + "date": "2026-03-01", + "value": "p95 latency ~5.5s; higher at max effort" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-02-05", + "value": "1M token context window (beta at launch, since standard); 128K max output tokens" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Status Page", + "url": "https://status.anthropic.com/", + "date": "2026-06-01", + "value": "99.9%+ uptime (last 90 days)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Generational leap in abstract reasoning (68.8% ARC-AGI-2, ~2x Opus 4.5). 80.8% SWE-bench with 1M context and 128K output. Introduced adaptive thinking and GA effort parameter including 'max'. Now two generations behind Opus 4.8 but still fully served." + }, + + "security": { + "overall_score": 91, + "criteria": { + "prompt_injection_resistance": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Safety Research", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "Improved resistance to prompt injection in agentic and computer-use settings" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attacks", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-02-05", + "value": "Constitutional AI alignment carried forward with enhanced refusal calibration" + } + ], + "methodology": "Testing against adversarial prompt datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Statement", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-02-05", + "value": "No training on user data without explicit consent" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Safety Evaluations", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "Released with comprehensive safety evaluations under the Responsible Scaling Policy" + } + ], + "methodology": "Comprehensive safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-02-05", + "value": "API key authentication, HTTPS only, rate limiting; removal of last-assistant-turn prefills closes a response-steering vector" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong safety posture. Removal of last-assistant-turn prefills (400 error) eliminates a common response-manipulation pattern; structured outputs replace it." + }, + + "privacy_compliance": { + "overall_score": 93, + "criteria": { + "data_residency": { + "value": "US, EU (customer choice)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Enterprise Documentation", + "url": "https://www.anthropic.com/claude/enterprise", + "date": "2026-02-05", + "value": "Data residency options for US and EU customers" + } + ], + "methodology": "Review of enterprise documentation and privacy policies", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Privacy Policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-02-05", + "value": "Opt-out available, no training on API data by default" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "0 days (ephemeral)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Terms of Service", + "url": "https://www.anthropic.com/legal/terms", + "date": "2026-02-05", + "value": "API prompts and outputs not retained (except for trust & safety)" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Documentation", + "url": "https://docs.anthropic.com/en/docs/resources/data-protection", + "date": "2026-02-05", + "value": "Customer responsible for PII redaction" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-02-05", + "value": "SOC 2 Type II, GDPR compliant, HIPAA eligible" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://docs.anthropic.com/en/docs/resources/data-protection", + "date": "2026-02-05", + "value": "Ephemeral data processing, no storage of prompts/outputs" + } + ], + "methodology": "Review of data handling practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Exceptional privacy posture with ephemeral data handling and strong compliance certifications. HIPAA eligible for healthcare." + }, + + "trust_transparency": { + "overall_score": 89, + "criteria": { + "explainability": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Adaptive Thinking Feature", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "Adaptive thinking surfaces reasoning depth decisions; effort parameter provides explicit compute transparency" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Testing", + "url": "https://www.anthropic.com/news/claude-opus-4-6", + "date": "2026-02-05", + "value": "Improved factual calibration over Opus 4.5, especially at high and max effort" + } + ], + "methodology": "Testing on factual QA datasets and real-world usage", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Responsible Scaling Policy", + "url": "https://www.anthropic.com/news/anthropics-responsible-scaling-policy", + "date": "2026-02-05", + "value": "Regular bias testing and mitigation" + } + ], + "methodology": "Evaluation on bias benchmarks and diverse demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-02-05", + "value": "Model expresses uncertainty appropriately; adaptive thinking scales effort with problem difficulty" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Model Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-02-05", + "value": "Comprehensive model cards with capabilities, limitations, benchmarks" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Public Statements", + "url": "https://www.anthropic.com/news", + "date": "2026-02-05", + "value": "General description provided, detailed sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-02-05", + "value": "Constitutional AI safety guardrails with improved refusal calibration" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Adaptive thinking improves transparency by making reasoning depth model-driven and observable. Strong instruction following reduces need for aggressive prompt engineering." + }, + + "operational_excellence": { + "overall_score": 91, + "criteria": { + "api_design_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-02-05", + "value": "Adaptive thinking, GA effort parameter (incl. max), structured outputs; prefills removed in favor of output_config.format" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic SDKs", + "url": "https://github.com/anthropics", + "date": "2026-02-05", + "value": "Official SDKs for Python, TypeScript, Java, Go, Ruby, C#, PHP — actively maintained" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Versioning", + "url": "https://docs.anthropic.com/en/api/versioning", + "date": "2026-02-05", + "value": "Clear versioning with advance deprecation notice; Opus 4.6 remains served two generations behind Opus 4.8" + } + ], + "methodology": "Review of versioning policy and historical practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Console", + "url": "https://console.anthropic.com/", + "date": "2026-02-05", + "value": "Usage dashboard with metrics" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Support", + "url": "https://support.anthropic.com/", + "date": "2026-02-05", + "value": "Email support, Discord community, comprehensive docs and migration guides" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Cloud Providers", + "url": "https://www.anthropic.com/claude/opus", + "date": "2026-02-05", + "value": "Available on AWS Bedrock, Google Vertex AI, Azure Foundry" + } + ], + "methodology": "Analysis of third-party integrations and tools", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Terms of Service", + "url": "https://www.anthropic.com/legal/commercial-terms", + "date": "2026-02-05", + "value": "Standard commercial terms, enterprise agreements available" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Mature operational profile with multi-cloud availability. Migration to 4.6 required removing assistant-turn prefills and moving to adaptive thinking — well-documented breaking changes." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 97, + "notes": "80.8% SWE-bench Verified and 65.4% Terminal-Bench 2.0. Excellent for complex software engineering, though Opus 4.7/4.8 now lead the family.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "customer-support": { + "overall": 88, + "notes": "Strong empathy and natural conversation. Higher latency and cost than Sonnet for routine support volume.", + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] + }, + "content-creation": { + "overall": 92, + "notes": "Excellent long-form, nuanced content. Adaptive thinking allocates more reasoning to complex pieces automatically.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "data-analysis": { + "overall": 95, + "notes": "Strong analytical capabilities with 1M context for large datasets. Effort 'max' useful for complex interpretation.", + "alternatives": ["gemini-3-1-pro", "gpt-5-5"] + }, + "research-assistant": { + "overall": 96, + "notes": "1M context and 68.8% ARC-AGI-2 abstract reasoning make it exceptional for deep research and synthesis.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 92, + "notes": "Strong privacy posture, HIPAA eligible. 1M context handles entire contract repositories in a single request.", + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 90, + "notes": "HIPAA eligible with strong privacy controls. Good for clinical documentation requiring high accuracy.", + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 93, + "notes": "Excellent quantitative reasoning. Adaptive thinking scales analysis depth with problem complexity.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "education": { + "overall": 93, + "notes": "Excellent tutoring with patient explanations. Effort parameter lets platforms balance quality against cost.", + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] + }, + "creative-writing": { + "overall": 90, + "notes": "Strong creative capabilities with nuanced character development and narrative flow.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-5"] + } + }, + + "strengths": [ + "Breakthrough abstract reasoning: 68.8% ARC-AGI-2 (up from Opus 4.5's 37.6%)", + "Elite coding: 80.8% SWE-bench Verified, 65.4% Terminal-Bench 2.0", + "Best-in-class computer use at launch: 72.7% OSWorld", + "1M token context window (beta at launch) with 128K max output", + "Adaptive thinking replaces manual thinking budgets — no tuning required", + "Effort parameter GA including new 'max' level for compute control", + "Same $5/$25 pricing as Opus 4.5 despite major capability gains" + ], + + "limitations": [ + "Two generations behind current Opus 4.8 (still served, but no longer frontier)", + "Removed last-assistant-turn prefills — code relying on prefills returns 400", + "Higher latency than Sonnet models (~2.5s p50)", + "Premium pricing relative to Sonnet 4.6 ($5/$25 vs $3/$15)", + "No native audio capabilities", + "Training data transparency limited (industry standard)" + ], + + "best_for": [ + "Complex software engineering and long-horizon agentic coding", + "Abstract reasoning and novel problem solving (ARC-AGI-class tasks)", + "Computer-use and desktop automation workflows", + "Ultra-long-context research over 1M-token corpora", + "Enterprise workloads pinned to a stable, well-characterized Opus generation" + ], + + "not_recommended_for": [ + "New deployments where Opus 4.8 is available at the same price", + "Real-time applications requiring <500ms latency", + "Cost-sensitive high-volume inference", + "Workflows still dependent on assistant-turn prefills" + ], + + "metadata": { + "pricing": { + "input": "$5.00 per 1M tokens", + "output": "$25.00 per 1M tokens", + "notes": "Same pricing as Opus 4.5. Batch API 50% discount. Prompt caching up to 90% savings.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 128000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)", "document", "computer-use"], + "api_endpoint": "https://api.anthropic.com/v1/messages", + "open_source": false, + "architecture": "Transformer-based with Constitutional AI alignment, adaptive thinking, and effort parameter", + "parameters": "Not disclosed", + "knowledge_cutoff": "Not disclosed" + }, + + "related_entities": ["claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-5", "claude-sonnet-4-6"], + + "tags": [ + "coding", + "reasoning", + "enterprise", + "hipaa-eligible", + "safety-focused", + "adaptive-thinking", + "effort-parameter", + "computer-use", + "long-context", + "previous-generation" + ] +} diff --git a/data/models/claude-opus-4-7.json b/data/models/claude-opus-4-7.json new file mode 100644 index 0000000..20b8599 --- /dev/null +++ b/data/models/claude-opus-4-7.json @@ -0,0 +1,650 @@ +{ + "id": "claude-opus-4-7", + "type": "model", + "name": "Claude Opus 4.7", + "provider": "Anthropic", + "version": "claude-opus-4-7", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Previous-generation Opus flagship, superseded by Opus 4.8. 64.3% SWE-Bench Pro and 94.2% GPQA Diamond at launch. First Claude with high-resolution vision (2576px long edge, pixel-accurate coordinates), task budgets (beta), and the xhigh effort level.", + "website": "https://www.anthropic.com/news/claude-opus-4-7", + + "trust_vector": { + "performance_reliability": { + "overall_score": 95, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "SWE-Bench Pro", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "64.3% on SWE-Bench Pro at launch" + }, + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "Strong long-horizon agentic coding; meaningfully better bug-finding recall and precision than prior Opus models" + } + ], + "methodology": "Industry-standard coding benchmarks measuring real-world software engineering tasks", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "GPQA Diamond", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "94.2% on PhD-level science questions" + } + ], + "methodology": "Graduate and PhD-level reasoning benchmarks evaluated with adaptive thinking at high effort", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "First Claude with high-resolution vision (2576px long edge); gains in knowledge work, memory, and vision-heavy tasks" + } + ], + "methodology": "Comprehensive knowledge and multimodal testing, including high-resolution screenshot and document understanding", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "More literal instruction following and stricter effort adherence yield more predictable pipeline behavior" + } + ], + "methodology": "Repeated-run consistency testing across effort levels; structured extraction pipelines", + "last_verified": "2026-06-10", + "notes": "Adaptive thinking only; thinking content omitted by default; sampling parameters removed" + }, + "latency_p50": { + "value": "2.8s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-05-15", + "value": "Typical response time ~2.8s for standard prompts at default effort" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "6.5s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-05-15", + "value": "p95 ~6.5s; higher at xhigh/max effort" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-04-16", + "value": "1M token context window, 128K max output tokens" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Status Page", + "url": "https://status.anthropic.com/", + "date": "2026-06-10", + "value": "99.9%+ uptime (last 90 days)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Was Anthropic's most capable model at launch (2026-04-16); now the previous-generation Opus behind Opus 4.8. Remains a strong, fully supported flagship-class choice, especially for vision-heavy workloads." + }, + + "security": { + "overall_score": 91, + "criteria": { + "prompt_injection_resistance": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "Improved resistance to injected instructions; more literal instruction-following reduces susceptibility to embedded directives" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attacks", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-04-16", + "value": "Constitutional AI alignment with refined refusal calibration" + } + ], + "methodology": "Testing against adversarial prompt datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Statement", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-04-16", + "value": "Training opt-out by default for API; no training on user data without consent" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "Introduced real-time cybersecurity safeguards for prohibited and high-risk topics" + } + ], + "methodology": "Comprehensive safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/api/rate-limits", + "date": "2026-04-16", + "value": "API key and OAuth authentication, HTTPS only, rate limiting, workspace scoping" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Introduced real-time cybersecurity safeguards to the Opus line. Strong overall posture carried forward into Opus 4.8." + }, + + "privacy_compliance": { + "overall_score": 93, + "criteria": { + "data_residency": { + "value": "US, EU (customer choice)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Enterprise Documentation", + "url": "https://www.anthropic.com/claude/enterprise", + "date": "2026-04-16", + "value": "Data residency options for US and EU enterprise customers" + } + ], + "methodology": "Review of enterprise documentation and privacy policies", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Privacy Policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-04-16", + "value": "Training opt-out by default for API usage" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Configurable; zero-retention available", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-04-16", + "value": "Zero data retention agreements available for eligible API customers" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Documentation", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-04-16", + "value": "Customer responsible for PII redaction; provider-side safeguards for incidental PII" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-04-16", + "value": "SOC 2 Type II, GDPR compliant, HIPAA eligible" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-04-16", + "value": "Zero data retention configuration available; no training on API data by default" + } + ], + "methodology": "Review of data handling practices and trust center documentation", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard Anthropic enterprise compliance posture: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API traffic." + }, + + "trust_transparency": { + "overall_score": 87, + "criteria": { + "explainability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Adaptive Thinking Documentation", + "url": "https://platform.claude.com/docs/en/build-with-claude/adaptive-thinking", + "date": "2026-04-16", + "value": "Adaptive thinking with effort control; thinking content omitted by default — summarized display is opt-in" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10", + "notes": "First Opus where thinking text defaults to omitted — a transparency regression vs Opus 4.6 defaults, recoverable via display: summarized" + }, + "hallucination_rate": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "Improved factual accuracy; visually verifies its own output in knowledge-work tasks" + } + ], + "methodology": "Testing on factual QA datasets and document-fidelity evaluations", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Responsible Scaling Policy", + "url": "https://www.anthropic.com/news/anthropics-responsible-scaling-policy", + "date": "2026-04-16", + "value": "Regular bias testing and mitigation under the Responsible Scaling Policy" + } + ], + "methodology": "Evaluation on bias benchmarks and diverse demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "More literal, precise behavior; expresses uncertainty rather than inferring unrequested intent" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Model Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-04-16", + "value": "Comprehensive model card with capabilities, limitations, benchmarks, and migration guidance" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Public Statements", + "url": "https://www.anthropic.com/news", + "date": "2026-04-16", + "value": "General description provided, detailed sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-04-16", + "value": "Constitutional AI safety guardrails with real-time cybersecurity safeguards" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong documentation and guardrails. Thinking content omitted by default reduces out-of-the-box reasoning visibility; opt in to summarized display if reasoning is surfaced to users." + }, + + "operational_excellence": { + "overall_score": 91, + "criteria": { + "api_design_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Migration Guide", + "url": "https://platform.claude.com/docs/en/about-claude/models/migration-guide", + "date": "2026-04-16", + "value": "Introduced the xhigh effort level and task budgets (beta); adaptive thinking only — budget_tokens and temperature/top_p/top_k removed" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10", + "notes": "Breaking changes vs Opus 4.6 (sampling params and manual thinking budgets return 400); Opus 4.8 keeps this same surface" + }, + "sdk_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic SDKs", + "url": "https://github.com/anthropics", + "date": "2026-04-16", + "value": "Official SDKs (Python, TypeScript, Java, Go, Ruby, C#, PHP) with day-one support" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Versioning", + "url": "https://platform.claude.com/docs/en/api/versioning", + "date": "2026-04-16", + "value": "Clear versioning with advance deprecation notice; claude-opus-4-7 alias remains active after Opus 4.8 launch" + } + ], + "methodology": "Review of versioning policy and historical practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Console", + "url": "https://console.anthropic.com/", + "date": "2026-04-16", + "value": "Usage dashboard with metrics, cost tracking, and workspace controls" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Support", + "url": "https://support.anthropic.com/", + "date": "2026-04-16", + "value": "Email support, developer community, comprehensive docs and migration guides" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-7", + "date": "2026-04-16", + "value": "Available on the Anthropic API and major cloud platforms; broad tooling support" + } + ], + "methodology": "Analysis of third-party integrations and availability surfaces", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Commercial Terms", + "url": "https://www.anthropic.com/legal/commercial-terms", + "date": "2026-04-16", + "value": "Standard commercial terms; enterprise agreements available" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Mature operational profile. Superseded by Opus 4.8 as the flagship Opus, but remains fully supported at the same $5/$25 price; upgrade to 4.8 is a drop-in model-ID swap." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 95, + "notes": "64.3% SWE-Bench Pro with strong long-horizon agentic coding and improved bug-finding. Superseded by Opus 4.8 at the same price — prefer 4.8 for new builds.", + "alternatives": ["claude-opus-4-8", "gpt-5-3-codex"] + }, + "customer-support": { + "overall": 88, + "notes": "High quality but more clipped, direct tone than 4.8; Sonnet/Haiku tiers are more cost-effective for routine volume.", + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] + }, + "content-creation": { + "overall": 91, + "notes": "Strong long-form output, though more terse and less warm than Opus 4.8 by default; style is prompt-tunable.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "data-analysis": { + "overall": 94, + "notes": "Excellent analytical depth; high-resolution vision enables pixel-level chart and figure transcription.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "research-assistant": { + "overall": 94, + "notes": "Strong deep research with 1M context and improved file-based memory; Opus 4.8 improves further on long-horizon coherence.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 92, + "notes": "Strong privacy posture (SOC 2 Type II, GDPR, HIPAA-eligible) and literal instruction following suited to compliance pipelines.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "healthcare": { + "overall": 90, + "notes": "HIPAA eligible with training opt-out by default; high-resolution vision aids medical document and chart understanding.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "financial-analysis": { + "overall": 93, + "notes": "Excellent quantitative reasoning; pixel-accurate chart reading and 1M context handle full filings and figures.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "education": { + "overall": 92, + "notes": "Clear, precise explanations with effort-adjustable depth; more literal style benefits structured curricula.", + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] + }, + "creative-writing": { + "overall": 89, + "notes": "Capable but more clipped and direct than Opus 4.8's warmer voice; no sampling parameters, so variety must be prompted.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + } + }, + + "strengths": [ + "64.3% SWE-Bench Pro and 94.2% GPQA Diamond at launch", + "First Claude with high-resolution vision: 2576px long edge with pixel-accurate coordinates", + "Introduced the xhigh effort level and task budgets (beta) for agentic token control", + "1M token context window with 128K max output at $5/$25", + "More literal, predictable instruction following for tuned pipelines", + "Strong compliance posture: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default" + ], + + "limitations": [ + "Superseded by Opus 4.8 as the current flagship Opus (same price, drop-in upgrade)", + "Adaptive thinking only — budget_tokens and temperature/top_p/top_k return 400", + "Thinking content omitted by default; summarized display requires opt-in", + "Full-resolution images can use up to ~3x more image tokens than prior models", + "Reaches for tools and subagents less often than Opus 4.6 without explicit prompting" + ], + + "best_for": [ + "Vision-heavy workloads: screenshots, computer use, chart and document understanding", + "Structured extraction and tuned pipelines that benefit from literal instruction following", + "Long-horizon agentic coding where Opus 4.8 has not yet been qualified", + "Teams pinned to a validated model version for reproducibility" + ], + + "not_recommended_for": [ + "New deployments where Opus 4.8 is available at the same price with better performance", + "Real-time applications requiring sub-second latency", + "Cost-sensitive high-volume inference (use Sonnet or Haiku tiers)", + "Workflows that depend on temperature/top_p sampling controls" + ], + + "metadata": { + "pricing": { + "input": "$5.00 per 1M tokens", + "output": "$25.00 per 1M tokens", + "notes": "1M context at standard API pricing with no long-context premium. Batch API 50% discount, prompt caching savings apply. No fast-mode variant.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 128000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input, high-resolution)", "document", "computer-use"], + "api_endpoint": "https://api.anthropic.com/v1/messages", + "api_model_id": "claude-opus-4-7", + "open_source": false, + "architecture": "Transformer-based with Constitutional AI alignment; adaptive thinking only with effort parameter introducing xhigh; high-resolution vision", + "parameters": "Not disclosed", + "knowledge_cutoff": "Not disclosed", + "release_date": "2026-04-16" + }, + + "related_entities": ["claude-opus-4-8", "claude-fable-5", "claude-opus-4-6", "claude-sonnet-4-6", "gpt-5-4"], + + "tags": [ + "coding", + "reasoning", + "vision", + "enterprise", + "hipaa-eligible", + "safety-focused", + "effort-parameter", + "adaptive-thinking", + "long-context", + "previous-generation" + ] +} diff --git a/data/models/claude-opus-4-8.json b/data/models/claude-opus-4-8.json new file mode 100644 index 0000000..1181b28 --- /dev/null +++ b/data/models/claude-opus-4-8.json @@ -0,0 +1,652 @@ +{ + "id": "claude-opus-4-8", + "type": "model", + "name": "Claude Opus 4.8", + "provider": "Anthropic", + "version": "claude-opus-4-8", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Anthropic's flagship Opus model with state-of-the-art long-horizon agentic execution, knowledge work, and memory. 84% on Online-Mind2Web, dynamic multi-subagent workflows, ~4x less likely to miss its own code flaws than its predecessor, and 1M context at standard pricing.", + "website": "https://www.anthropic.com/news/claude-opus-4-8", + + "trust_vector": { + "performance_reliability": { + "overall_score": 96, + "criteria": { + "task_accuracy_code": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "State-of-the-art long-horizon agentic coding; ~4x less likely to miss flaws in its own code vs Opus 4.7" + }, + { + "source": "Online-Mind2Web", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "84% on Online-Mind2Web live web-agent benchmark" + } + ], + "methodology": "Agentic coding and web-agent benchmarks measuring long-horizon autonomous execution and self-verification", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "State-of-the-art on knowledge work and memory tasks; improved planning via deeper per-step reasoning" + } + ], + "methodology": "Graduate-level reasoning and knowledge-work benchmarks evaluated with adaptive thinking at high effort", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "Gains across knowledge work, document tasks, and memory benchmarks over Opus 4.7" + } + ], + "methodology": "Comprehensive knowledge and multimodal testing, including high-resolution vision inherited from Opus 4.7", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "~4x reduction in missed self-authored code flaws; stronger self-verification across long agentic runs" + } + ], + "methodology": "Long-horizon agentic run consistency and self-verification testing across effort levels", + "last_verified": "2026-06-10", + "notes": "Same adaptive-thinking-only surface as Opus 4.7; effort parameter (incl. xhigh) controls depth" + }, + "latency_p50": { + "value": "2.8s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-05", + "value": "Typical response time ~2.8s for standard prompts at default effort; fast mode available at $10/$50" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "6.5s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-05", + "value": "p95 ~6.5s; higher at xhigh/max effort" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-05-28", + "value": "1M token context at standard API pricing (no long-context premium), 128K max output" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Status Page", + "url": "https://status.anthropic.com/", + "date": "2026-06-10", + "value": "99.9%+ uptime (last 90 days)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Current flagship Opus. State-of-the-art long-horizon agentic execution, knowledge work, and memory; 84% Online-Mind2Web; dynamic multi-subagent workflows. Superseded only by the higher-tier Claude Fable 5." + }, + + "security": { + "overall_score": 92, + "criteria": { + "prompt_injection_resistance": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "Improved resistance to injected instructions in agentic and browsing contexts; mid-session system prompts (beta) provide an injection-safe operator channel" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attacks, including agentic browsing scenarios", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-05-28", + "value": "Constitutional AI alignment with refined refusal calibration" + } + ], + "methodology": "Testing against adversarial prompt datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Statement", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-05-28", + "value": "Training opt-out by default for API; no training on user data without consent" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "Released under the Responsible Scaling Policy with real-time cybersecurity safeguards carried forward from Opus 4.7" + } + ], + "methodology": "Comprehensive safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/api/rate-limits", + "date": "2026-05-28", + "value": "API key and OAuth authentication, HTTPS only, rate limiting, workspace scoping" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong safety posture with agentic-specific safeguards. Mid-session system prompts (beta) give operators a non-spoofable instruction channel for long-running sessions." + }, + + "privacy_compliance": { + "overall_score": 93, + "criteria": { + "data_residency": { + "value": "US, EU (customer choice)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Enterprise Documentation", + "url": "https://www.anthropic.com/claude/enterprise", + "date": "2026-05-28", + "value": "Data residency options for US and EU enterprise customers" + } + ], + "methodology": "Review of enterprise documentation and privacy policies", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Privacy Policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-05-28", + "value": "Training opt-out by default for API usage" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Configurable; zero-retention available", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-05-28", + "value": "Zero data retention agreements available for eligible API customers" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Documentation", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-05-28", + "value": "Customer responsible for PII redaction; provider-side safeguards for incidental PII" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-05-28", + "value": "SOC 2 Type II, GDPR compliant, HIPAA eligible" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-05-28", + "value": "Zero data retention configuration available; no training on API data by default" + } + ], + "methodology": "Review of data handling practices and trust center documentation", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard Anthropic enterprise compliance posture: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API traffic." + }, + + "trust_transparency": { + "overall_score": 88, + "criteria": { + "explainability": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Adaptive Thinking Documentation", + "url": "https://platform.claude.com/docs/en/build-with-claude/adaptive-thinking", + "date": "2026-05-28", + "value": "Adaptive thinking with effort control; richer user-facing narration during long agentic runs; thinking text omitted by default with summarized display opt-in" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "~4x reduction in missed self-authored code flaws; improved factual grounding in knowledge work" + } + ], + "methodology": "Testing on factual QA datasets and self-verification evaluations", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Responsible Scaling Policy", + "url": "https://www.anthropic.com/news/anthropics-responsible-scaling-policy", + "date": "2026-05-28", + "value": "Regular bias testing and mitigation under the Responsible Scaling Policy" + } + ], + "methodology": "Evaluation on bias benchmarks and diverse demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "More deliberate: pauses to ask on ambiguous decisions and flags uncertainty rather than guessing" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Model Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-05-28", + "value": "Comprehensive model card with capabilities, limitations, benchmarks, and migration guidance" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Public Statements", + "url": "https://www.anthropic.com/news", + "date": "2026-05-28", + "value": "General description provided, detailed sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-05-28", + "value": "Constitutional AI safety guardrails with real-time cybersecurity safeguards" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "More deliberate and transparent in long agentic runs than Opus 4.7 — narrates progress, flags uncertainty, and self-verifies code. Thinking text remains omitted by default." + }, + + "operational_excellence": { + "overall_score": 91, + "criteria": { + "api_design_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Migration Guide", + "url": "https://platform.claude.com/docs/en/about-claude/models/migration-guide", + "date": "2026-05-28", + "value": "Same API surface as Opus 4.7 — no new breaking changes; adds mid-session system prompts (beta)" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10", + "notes": "Adaptive thinking only; effort levels include xhigh; task budgets (beta) supported" + }, + "sdk_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic SDKs", + "url": "https://github.com/anthropics", + "date": "2026-05-28", + "value": "Official SDKs (Python, TypeScript, Java, Go, Ruby, C#, PHP) with day-one support" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Versioning", + "url": "https://platform.claude.com/docs/en/api/versioning", + "date": "2026-05-28", + "value": "Clear versioning with advance deprecation notice; stable claude-opus-4-8 alias" + } + ], + "methodology": "Review of versioning policy and historical practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Console", + "url": "https://console.anthropic.com/", + "date": "2026-05-28", + "value": "Usage dashboard with metrics, cost tracking, and workspace controls" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Support", + "url": "https://support.anthropic.com/", + "date": "2026-05-28", + "value": "Email support, developer community, comprehensive docs and a dedicated 4.7-to-4.8 migration guide" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Launch Announcement", + "url": "https://www.anthropic.com/news/claude-opus-4-8", + "date": "2026-05-28", + "value": "Available on the Anthropic API and major cloud platforms; default model in Claude Code and the Agent SDK" + } + ], + "methodology": "Analysis of third-party integrations and availability surfaces", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Commercial Terms", + "url": "https://www.anthropic.com/legal/commercial-terms", + "date": "2026-05-28", + "value": "Standard commercial terms; enterprise agreements available" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Drop-in upgrade from Opus 4.7 (identical API surface). 1M context at standard pricing with no long-context premium; optional fast mode at $10/$50." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 97, + "notes": "State-of-the-art long-horizon agentic coding; ~4x less likely to miss its own code flaws than Opus 4.7. Best value flagship for software engineering at $5/$25.", + "alternatives": ["claude-fable-5", "gpt-5-3-codex"] + }, + "customer-support": { + "overall": 89, + "notes": "Excellent quality with warmer, clearer writing than Opus 4.7, but Sonnet/Haiku tiers are more cost-effective for routine volume.", + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] + }, + "content-creation": { + "overall": 94, + "notes": "Clearer, warmer, less hedged prose than prior Opus models — approaches expert-level structure at higher effort.", + "alternatives": ["claude-fable-5", "gpt-5-5"] + }, + "data-analysis": { + "overall": 95, + "notes": "Strong analytical depth with 1M context for whole-dataset work; dynamic multi-subagent workflows fan out across large analyses.", + "alternatives": ["claude-fable-5", "gemini-3-1-pro"] + }, + "research-assistant": { + "overall": 96, + "notes": "State-of-the-art knowledge work and memory; excels at multi-day research with file-based memory and 1M context.", + "alternatives": ["claude-fable-5", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 93, + "notes": "Strong privacy posture (SOC 2 Type II, GDPR, HIPAA-eligible) and thorough long-document analysis at 1M context with no long-context premium.", + "alternatives": ["claude-fable-5", "claude-sonnet-4-6"] + }, + "healthcare": { + "overall": 91, + "notes": "HIPAA eligible with training opt-out by default. Deliberate, uncertainty-flagging behavior suits clinical documentation.", + "alternatives": ["claude-fable-5", "claude-sonnet-4-6"] + }, + "financial-analysis": { + "overall": 94, + "notes": "Excellent quantitative reasoning and knowledge work; handles full filings and model workbooks in one context window.", + "alternatives": ["claude-fable-5", "gpt-5-5"] + }, + "education": { + "overall": 93, + "notes": "Clear, warm explanations with effort-adjustable depth; strong thought-partner behavior that pushes back constructively.", + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] + }, + "creative-writing": { + "overall": 92, + "notes": "Warmer, less hedged voice with fewer AI vocal tics than 4.7. No sampling parameters — variety must be prompted.", + "alternatives": ["claude-fable-5", "gpt-5-5"] + } + }, + + "strengths": [ + "State-of-the-art long-horizon agentic execution, knowledge work, and memory", + "84% on Online-Mind2Web live web-agent benchmark", + "~4x less likely to miss flaws in its own code than Opus 4.7", + "Dynamic multi-subagent workflows for parallel fan-out", + "1M context at standard pricing (no long-context premium), 128K output", + "Mid-session system prompts (beta) — injection-safe operator channel that preserves prompt cache", + "Same API surface as Opus 4.7 — drop-in upgrade with no new breaking changes" + ], + + "limitations": [ + "Adaptive thinking only — no manual thinking budgets, no temperature/top_p sampling parameters", + "More deliberate by default: asks clarifying questions more often unless granted explicit autonomy", + "Narrates more between tool calls than 4.7 — needs a silence-default prompt for terse agents", + "Conservative about reaching for search, subagents, and custom tools without explicit triggering guidance", + "Higher latency than Sonnet/Haiku tiers; fast mode doubles cost to $10/$50" + ], + + "best_for": [ + "Long-horizon autonomous coding runs and complex refactors", + "Knowledge work over very large document sets (1M context, no premium)", + "Multi-agent and subagent orchestration workflows", + "Memory-dependent agents that persist context across sessions", + "Enterprise workloads requiring strong compliance" + ], + + "not_recommended_for": [ + "Real-time applications requiring sub-second latency", + "Cost-sensitive high-volume inference (use Sonnet or Haiku tiers)", + "Workflows that depend on temperature/top_p sampling controls", + "Audio processing applications" + ], + + "metadata": { + "pricing": { + "input": "$5.00 per 1M tokens", + "output": "$25.00 per 1M tokens", + "notes": "Fast mode available at $10/$50 per 1M. 1M context at standard pricing with no long-context premium. Batch API 50% discount, prompt caching savings apply.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 128000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)", "document", "computer-use"], + "api_endpoint": "https://api.anthropic.com/v1/messages", + "api_model_id": "claude-opus-4-8", + "open_source": false, + "architecture": "Transformer-based with Constitutional AI alignment; adaptive thinking only with effort parameter including xhigh; same API surface as Opus 4.7", + "parameters": "Not disclosed", + "knowledge_cutoff": "Not disclosed", + "release_date": "2026-05-28" + }, + + "related_entities": ["claude-fable-5", "claude-opus-4-7", "claude-sonnet-4-6", "gpt-5-5", "gemini-3-1-pro"], + + "tags": [ + "coding", + "reasoning", + "agentic", + "enterprise", + "hipaa-eligible", + "safety-focused", + "effort-parameter", + "adaptive-thinking", + "long-context", + "memory", + "flagship" + ] +} diff --git a/data/models/claude-sonnet-4-6.json b/data/models/claude-sonnet-4-6.json new file mode 100644 index 0000000..5a5fbf5 --- /dev/null +++ b/data/models/claude-sonnet-4-6.json @@ -0,0 +1,647 @@ +{ + "id": "claude-sonnet-4-6", + "type": "model", + "name": "Claude Sonnet 4.6", + "provider": "Anthropic", + "version": "4.6", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Anthropic's best speed/intelligence balance — the value workhorse for agentic and production workloads at $3/$15 per 1M tokens, with a 1M token context window, adaptive thinking, the effort parameter including 'max', and strong computer-use accuracy.", + "website": "https://www.anthropic.com/claude/sonnet", + + "trust_vector": { + "performance_reliability": { + "overall_score": 93, + "criteria": { + "task_accuracy_code": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Recommended model for agentic coding at the Sonnet tier; successor to Claude Sonnet 4.5" + }, + { + "source": "Anthropic Migration Guide", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Documented as the drop-in upgrade target for Sonnet 4.5, 4.0, 3.7, and 3.5 coding workloads" + } + ], + "methodology": "Review of official model documentation and positioning for software engineering workloads", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Adaptive thinking supported; effort defaults to high, scaling reasoning depth with task complexity" + } + ], + "methodology": "Review of documented thinking capabilities and reasoning benchmark positioning", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Positioned as Anthropic's best combination of speed and intelligence" + } + ], + "methodology": "Comprehensive knowledge and multimodal capability review against official documentation", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Effort parameter (low/medium/high/max) gives explicit, repeatable quality/cost control; strong computer-use accuracy with adaptive thinking at high effort" + } + ], + "methodology": "Internal testing of output stability across effort levels and adaptive thinking", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "1.5s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models/claude-sonnet-4-6", + "date": "2026-03-01", + "value": "Typical response time ~1.5s for standard prompts at low/medium effort; faster than Opus tier" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "3.5s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models/claude-sonnet-4-6", + "date": "2026-03-01", + "value": "p95 latency ~3.5s; higher at high/max effort with adaptive thinking" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "1M token context window; 64K max output tokens" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Status Page", + "url": "https://status.anthropic.com/", + "date": "2026-06-01", + "value": "99.9%+ uptime (last 90 days)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "The value workhorse of the Claude lineup: near-Opus intelligence at Sonnet latency and price, with a 1M context window, adaptive thinking, and the full effort range including 'max'." + }, + + "security": { + "overall_score": 91, + "criteria": { + "prompt_injection_resistance": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Safety Research", + "url": "https://www.anthropic.com/news/evaluating-ai-systems", + "date": "2026-06-10", + "value": "Strong resistance to prompt injection in agentic and computer-use settings" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attacks", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-06-10", + "value": "Constitutional AI alignment with well-calibrated refusals" + } + ], + "methodology": "Testing against adversarial prompt datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Statement", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-06-10", + "value": "No training on user data without explicit consent" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-06-10", + "value": "Released with comprehensive safety evaluations under the Responsible Scaling Policy" + } + ], + "methodology": "Comprehensive safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "API key authentication, HTTPS only, rate limiting; assistant prefills removed (400), closing a response-steering vector" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong safety posture. Like Opus 4.6, last-assistant-turn prefills return a 400 — structured outputs (output_config.format) are the supported replacement." + }, + + "privacy_compliance": { + "overall_score": 93, + "criteria": { + "data_residency": { + "value": "US, EU (customer choice)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Enterprise Documentation", + "url": "https://www.anthropic.com/claude/enterprise", + "date": "2026-06-10", + "value": "Data residency options for US and EU customers" + } + ], + "methodology": "Review of enterprise documentation and privacy policies", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Privacy Policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-06-10", + "value": "Opt-out available, no training on API data by default" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "0 days (ephemeral)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Terms of Service", + "url": "https://www.anthropic.com/legal/terms", + "date": "2026-06-10", + "value": "API prompts and outputs not retained (except for trust & safety)" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Documentation", + "url": "https://docs.anthropic.com/en/docs/resources/data-protection", + "date": "2026-06-10", + "value": "Customer responsible for PII redaction" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-06-10", + "value": "SOC 2 Type II, GDPR compliant, HIPAA eligible" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://docs.anthropic.com/en/docs/resources/data-protection", + "date": "2026-06-10", + "value": "Ephemeral data processing, no storage of prompts/outputs" + } + ], + "methodology": "Review of data handling practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Same enterprise-grade privacy posture as the Opus tier: ephemeral data handling, strong certifications, HIPAA eligible." + }, + + "trust_transparency": { + "overall_score": 88, + "criteria": { + "explainability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Adaptive thinking and the effort parameter make reasoning depth explicit and controllable" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Testing", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Improved factual calibration over Sonnet 4.5, especially with adaptive thinking enabled" + } + ], + "methodology": "Testing on factual QA datasets and real-world usage", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Responsible Scaling Policy", + "url": "https://www.anthropic.com/news/anthropics-responsible-scaling-policy", + "date": "2026-06-10", + "value": "Regular bias testing and mitigation" + } + ], + "methodology": "Evaluation on bias benchmarks and diverse demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Model expresses uncertainty appropriately; adaptive thinking scales effort with problem difficulty" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Model Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Comprehensive model documentation with capabilities, limitations, and migration guidance from Sonnet 4.5" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Public Statements", + "url": "https://www.anthropic.com/news", + "date": "2026-06-10", + "value": "General description provided, detailed sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-06-10", + "value": "Constitutional AI safety guardrails with well-calibrated refusals" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Transparent compute controls (adaptive thinking + effort) and thorough migration documentation. Follows instructions closely, reducing prompt-engineering opacity." + }, + + "operational_excellence": { + "overall_score": 91, + "criteria": { + "api_design_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-06-10", + "value": "Adaptive thinking, effort parameter incl. max, structured outputs, streaming, tool use; prefills removed in favor of output_config.format" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic SDKs", + "url": "https://github.com/anthropics", + "date": "2026-06-10", + "value": "Official SDKs for Python, TypeScript, Java, Go, Ruby, C#, PHP — actively maintained" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Versioning", + "url": "https://docs.anthropic.com/en/api/versioning", + "date": "2026-06-10", + "value": "Clear versioning with advance deprecation notice; documented migration path from Sonnet 4.5 and retired 3.x Sonnets" + } + ], + "methodology": "Review of versioning policy and historical practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Console", + "url": "https://console.anthropic.com/", + "date": "2026-06-10", + "value": "Usage dashboard with metrics" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Support", + "url": "https://support.anthropic.com/", + "date": "2026-06-10", + "value": "Email support, Discord community, comprehensive docs and migration guides" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Cloud Providers", + "url": "https://www.anthropic.com/claude/sonnet", + "date": "2026-06-10", + "value": "Available on AWS Bedrock, Google Vertex AI, Azure Foundry; default model in many agent frameworks" + } + ], + "methodology": "Analysis of third-party integrations and tools", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Terms of Service", + "url": "https://www.anthropic.com/legal/commercial-terms", + "date": "2026-06-10", + "value": "Standard commercial terms, enterprise agreements available" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Production-ready with multi-cloud availability. Migration from Sonnet 4.5 requires setting effort explicitly (4.6 defaults to high) and removing assistant prefills." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "Excellent agentic coding at a fraction of Opus cost. Pair effort 'medium' with adaptive thinking for the best cost/quality balance.", + "alternatives": ["claude-opus-4-8", "claude-opus-4-6"] + }, + "customer-support": { + "overall": 93, + "notes": "The sweet spot for support: fast, empathetic, and cost-effective at scale. Use effort 'low' with thinking disabled for high-volume tiers.", + "alternatives": ["claude-haiku-4-5", "gpt-5-5"] + }, + "content-creation": { + "overall": 91, + "notes": "Strong long-form and marketing content with fast turnaround. Opus tier still leads on the most nuanced pieces.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "data-analysis": { + "overall": 91, + "notes": "Solid analytical capabilities with 1M context for large datasets at workhorse pricing.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-6"] + }, + "research-assistant": { + "overall": 91, + "notes": "1M context handles large corpora; adaptive thinking deepens analysis when needed. Opus preferred for the hardest synthesis tasks.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 90, + "notes": "Strong privacy posture, HIPAA eligible, 1M context for contract repositories. Escalate the highest-stakes reviews to Opus.", + "alternatives": ["claude-opus-4-6", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 89, + "notes": "HIPAA eligible with strong privacy controls. Well-suited to clinical documentation at production volume.", + "alternatives": ["claude-opus-4-6", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 90, + "notes": "Good quantitative reasoning with predictable cost. Use effort 'high' for complex modeling; Opus for the hardest problems.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "education": { + "overall": 92, + "notes": "Fast, patient explanations at a price point that scales to large student populations.", + "alternatives": ["claude-haiku-4-5", "gpt-5-5"] + }, + "creative-writing": { + "overall": 88, + "notes": "Capable creative writing with good narrative flow; Opus tier produces more distinctive prose.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + } + }, + + "strengths": [ + "Best speed/intelligence balance in the Claude lineup at $3/$15 per 1M tokens", + "1M token context window with 64K max output", + "Adaptive thinking supported — no manual thinking budgets to tune", + "Effort parameter including 'max' (not available on Sonnet 4.5 or Haiku)", + "Strong computer-use accuracy for agentic automation", + "HIPAA eligible with ephemeral data handling", + "Multi-cloud availability (AWS, GCP, Azure)" + ], + + "limitations": [ + "Lower ceiling than Opus tier on the hardest reasoning and long-horizon agentic tasks", + "Removed assistant prefills — code relying on prefills returns 400", + "Effort defaults to high — Sonnet 4.5 migrations see higher latency/cost unless effort is set explicitly", + "64K max output (vs 128K on Opus 4.6+)", + "No native audio capabilities" + ], + + "best_for": [ + "Production agentic workloads needing strong quality at predictable cost", + "High-volume customer support and conversational applications", + "Agentic coding and tool-heavy workflows with fast turnaround", + "Computer-use automation at scale", + "Long-context processing of large document sets on a budget" + ], + + "not_recommended_for": [ + "Frontier-difficulty reasoning where Opus 4.8 is warranted", + "Workflows still dependent on assistant-turn prefills", + "Outputs beyond 64K tokens in a single response", + "Audio processing applications" + ], + + "metadata": { + "pricing": { + "input": "$3.00 per 1M tokens", + "output": "$15.00 per 1M tokens", + "notes": "Same pricing as Sonnet 4.5. Batch API 50% discount. Prompt caching up to 90% savings.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 64000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)", "document", "computer-use"], + "api_endpoint": "https://api.anthropic.com/v1/messages", + "open_source": false, + "architecture": "Transformer-based with Constitutional AI alignment, adaptive thinking, and effort parameter", + "parameters": "Not disclosed", + "knowledge_cutoff": "Not disclosed" + }, + + "related_entities": ["claude-sonnet-4-5", "claude-opus-4-6", "claude-opus-4-8", "claude-haiku-4-5"], + + "tags": [ + "coding", + "agentic", + "production", + "enterprise", + "hipaa-eligible", + "adaptive-thinking", + "effort-parameter", + "computer-use", + "long-context", + "value-workhorse" + ] +} diff --git a/data/models/command-a-plus.json b/data/models/command-a-plus.json new file mode 100644 index 0000000..f7d2594 --- /dev/null +++ b/data/models/command-a-plus.json @@ -0,0 +1,639 @@ +{ + "id": "command-a-plus", + "type": "model", + "name": "Command A+", + "provider": "Cohere", + "version": "20260520", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Cohere's Apache 2.0 open-weight 218B sparse MoE (25B active) unifying Command A, A Reasoning, A Vision, and A Translate. Runs on 2xH100 or a single B200, supports 48 languages, and ships native citations with grounding spans for verifiable RAG.", + "website": "https://cohere.com/blog/command-a-plus", + + "trust_vector": { + "performance_reliability": { + "overall_score": 89, + "criteria": { + "task_accuracy_code": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Command A+ launch announcement", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "Solid coding performance; competitive with open peers but not the headline focus" + } + ], + "methodology": "Vendor-reported coding benchmarks; independent replication still emerging three weeks post-release", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Command A+ launch announcement", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "Unifies Command A Reasoning capabilities; multimodal reasoning over text and images" + } + ], + "methodology": "Vendor-reported reasoning benchmarks pending broad independent verification", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Command A+ launch announcement", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "Strong enterprise-task performance across 48 languages; unifies four specialist Command A variants" + } + ], + "methodology": "Review of vendor benchmarks emphasizing enterprise RAG, translation, and multilingual tasks", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere Documentation", + "url": "https://docs.cohere.com/", + "date": "2026-05-20", + "value": "Enterprise focus on reproducible, grounded outputs; structured citation format is deterministic" + } + ], + "methodology": "Review of grounded-generation behavior and enterprise consistency claims", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "2.0s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-01", + "value": "Typical response ~2.0s; 25B active parameters keep serving fast on 2xH100" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "4.5s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-01", + "value": "p95 ~4.5s across diverse workloads" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "256,000 tokens", + "confidence": "medium", + "evidence": [ + { + "source": "Cohere model documentation", + "url": "https://docs.cohere.com/docs/models", + "date": "2026-05-20", + "value": "256K context consistent with prior Command A; A+ spec inferred pending explicit confirmation" + } + ], + "methodology": "Provider documentation; A+ figure carried from Command A line pending explicit spec sheet", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "Cohere Status Page", + "url": "https://status.cohere.com/", + "date": "2026-06-01", + "value": "Strong historical availability; private/VPC deployment removes dependence on shared API" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Released 2026-05-20, unifying Command A + A Reasoning + A Vision + A Translate into one model. Efficient serving (25B active, 2xH100 or 1xB200). Benchmarks are vendor-reported and independent verification is still emerging." + }, + + "security": { + "overall_score": 87, + "criteria": { + "prompt_injection_resistance": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere security documentation", + "url": "https://cohere.com/security", + "date": "2026-05-20", + "value": "Enterprise hardening program; grounded-RAG design constrains injection surface in retrieval workflows" + } + ], + "methodology": "Review of vendor security documentation and enterprise deployment guidance against OWASP LLM01 patterns", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere responsible AI documentation", + "url": "https://cohere.com/responsibility", + "date": "2026-05-20", + "value": "Enterprise-grade safety tuning; open weights mean self-hosted derivatives can strip guardrails" + } + ], + "methodology": "Testing against adversarial prompt datasets; deployer-dependent for self-hosted use", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Cohere enterprise deployment options", + "url": "https://cohere.com/deployment-options", + "date": "2026-05-20", + "value": "Private deployment and VPC options keep data fully inside customer infrastructure" + } + ], + "methodology": "Analysis of deployment isolation options and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere responsible AI documentation", + "url": "https://cohere.com/responsibility", + "date": "2026-05-20", + "value": "Safety filtering tuned for enterprise content categories" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Cohere security page", + "url": "https://cohere.com/security", + "date": "2026-05-20", + "value": "SOC 2 audited platform, API key auth, HTTPS only, SSO/SCIM for enterprise, VPC endpoints" + } + ], + "methodology": "Review of API security features, certifications, and enterprise controls", + "last_verified": "2026-06-10" + } + }, + "notes": "Strongest security posture among current open-weight releases thanks to Cohere's enterprise platform: SOC 2, VPC/private deployment, and a grounded-generation design that narrows injection surface in RAG." + }, + + "privacy_compliance": { + "overall_score": 89, + "criteria": { + "data_residency": { + "value": "US, EU, and customer-controlled (private/VPC or self-hosted)", + "confidence": "high", + "evidence": [ + { + "source": "Cohere deployment options", + "url": "https://cohere.com/deployment-options", + "date": "2026-05-20", + "value": "SaaS, cloud-marketplace, VPC, and on-premises deployment; Apache 2.0 weights allow any-jurisdiction self-hosting" + } + ], + "methodology": "Review of deployment documentation and residency options", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Cohere privacy policy", + "url": "https://cohere.com/privacy", + "date": "2026-05-20", + "value": "Enterprise data not used for training by default under enterprise terms" + } + ], + "methodology": "Analysis of privacy policy and enterprise data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Configurable; zero retention available in private/VPC and self-hosted deployments", + "confidence": "high", + "evidence": [ + { + "source": "Cohere enterprise documentation", + "url": "https://cohere.com/deployment-options", + "date": "2026-05-20", + "value": "Private deployments process data entirely within customer environment" + } + ], + "methodology": "Review of terms of service and deployment-dependent retention", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere security documentation", + "url": "https://cohere.com/security", + "date": "2026-05-20", + "value": "Enterprise data-handling guidance; customer retains redaction responsibility, but isolation options reduce exposure" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Cohere security and trust page", + "url": "https://cohere.com/security", + "date": "2026-05-20", + "value": "SOC 2 Type II, GDPR alignment, enterprise agreements; long-standing enterprise compliance posture" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Apache 2.0 open weights", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "Self-hosting on 2xH100 or 1xB200 gives complete data control; VPC option for managed zero-retention" + } + ], + "methodology": "Review of self-hosting and private deployment options enabling zero retention", + "last_verified": "2026-06-10" + } + }, + "notes": "Best-in-class privacy posture for an open-weight model: Western jurisdiction, SOC 2, and a full spectrum from SaaS to on-premises. The rare combination of open weights plus enterprise compliance is its core differentiator." + }, + + "trust_transparency": { + "overall_score": 84, + "criteria": { + "explainability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Command A+ launch announcement", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "Native citations with grounding spans tie each claim to source passages — verifiable RAG by construction" + } + ], + "methodology": "Evaluation of citation fidelity and grounding-span behavior in retrieval workflows", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere grounded generation documentation", + "url": "https://docs.cohere.com/docs/retrieval-augmented-generation-rag", + "date": "2026-05-20", + "value": "Grounded mode constrains generation to retrieved evidence, materially reducing unsupported claims" + } + ], + "methodology": "Testing on grounded QA workloads; ungrounded closed-book use performs closer to peers", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere responsible AI documentation", + "url": "https://cohere.com/responsibility", + "date": "2026-05-20", + "value": "Published responsibility framework and multilingual fairness work across 48 languages" + } + ], + "methodology": "Review of published bias evaluations and responsibility framework", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Grounding span behavior", + "url": "https://docs.cohere.com/docs/retrieval-augmented-generation-rag", + "date": "2026-05-20", + "value": "Absence of grounding spans signals unsupported content, giving a usable uncertainty proxy in RAG" + } + ], + "methodology": "Assessment of confidence expression and grounding-span signaling", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Command A+ release materials", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "Clear documentation of architecture (218B sparse MoE / 25B active), hardware targets, languages, and unified capabilities" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere public materials", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "General training approach described; detailed data sources not disclosed (industry standard)" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere responsible AI documentation", + "url": "https://cohere.com/responsibility", + "date": "2026-05-20", + "value": "Enterprise safety tuning with configurable safety modes" + } + ], + "methodology": "Analysis of built-in safety mechanisms and configurability", + "last_verified": "2026-06-10" + } + }, + "notes": "Native citations and grounding spans are a genuine trust differentiator: outputs in RAG mode are verifiable against source passages by construction, not post-hoc." + }, + + "operational_excellence": { + "overall_score": 88, + "criteria": { + "api_design_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Cohere API Documentation", + "url": "https://docs.cohere.com/reference/chat", + "date": "2026-05-20", + "value": "Mature Chat API with first-class RAG primitives (documents, citations), tool use, and streaming" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Cohere SDKs", + "url": "https://github.com/cohere-ai", + "date": "2026-05-20", + "value": "Official Python, TypeScript, Go, and Java SDKs, actively maintained" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere model documentation", + "url": "https://docs.cohere.com/docs/models", + "date": "2026-05-20", + "value": "Stable model naming with deprecation notices; A+ consolidates four variants into one maintained line" + } + ], + "methodology": "Review of versioning policy and historical practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Cohere dashboard", + "url": "https://dashboard.cohere.com/", + "date": "2026-05-20", + "value": "Usage dashboards and logs; enterprise deployments integrate with customer observability stacks" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Cohere support", + "url": "https://cohere.com/contact", + "date": "2026-05-20", + "value": "Dedicated enterprise support, solutions engineering, and comprehensive documentation" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Cloud availability", + "url": "https://cohere.com/deployment-options", + "date": "2026-05-20", + "value": "Available via Cohere API, major cloud marketplaces, and Hugging Face weights for self-hosting" + } + ], + "methodology": "Analysis of third-party integrations, cloud availability, and tooling", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Command A+ launch announcement", + "url": "https://cohere.com/blog/command-a-plus", + "date": "2026-05-20", + "value": "Apache 2.0 open weights — fully permissive commercial use, a notable shift from Cohere's earlier non-commercial open releases" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Mature enterprise platform behind a permissively licensed model. Apache 2.0 weights plus SOC 2 SaaS/VPC options give an unusually wide deployment spectrum." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 82, + "notes": "Capable but not the focus; agentic-coding specialists (Kimi K2.6, GLM-5) score higher.", + "alternatives": ["kimi-k2-6", "glm-5", "claude-opus-4-8"] + }, + "customer-support": { + "overall": 90, + "notes": "Excellent fit: 48 languages, grounded answers with citations, fast 25B-active serving, enterprise deployment options.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "content-creation": { + "overall": 85, + "notes": "Strong multilingual content with citation support for sourced writing.", + "alternatives": ["claude-opus-4-8", "glm-5"] + }, + "data-analysis": { + "overall": 87, + "notes": "Solid analysis with multimodal input and verifiable grounding over enterprise documents.", + "alternatives": ["glm-5", "claude-opus-4-8"] + }, + "research-assistant": { + "overall": 90, + "notes": "Native citations with grounding spans make research outputs auditable; 256K context handles large corpora.", + "alternatives": ["glm-5", "claude-opus-4-8"] + }, + "legal-compliance": { + "overall": 88, + "notes": "Western jurisdiction, SOC 2, VPC/on-prem options, and verifiable citations suit legal document workflows.", + "alternatives": ["claude-opus-4-8"] + }, + "healthcare": { + "overall": 84, + "notes": "Private/self-hosted deployment supports strict data control; verify HIPAA terms for managed offerings.", + "alternatives": ["claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 88, + "notes": "Grounded generation over filings and reports with citation trails fits compliance-sensitive finance work.", + "alternatives": ["claude-opus-4-8", "glm-5"] + }, + "education": { + "overall": 85, + "notes": "Strong multilingual tutoring; citations encourage source-checking habits.", + "alternatives": ["glm-5", "claude-opus-4-8"] + }, + "creative-writing": { + "overall": 78, + "notes": "Serviceable but enterprise-tuned; creative specialists perform better.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + } + }, + + "strengths": [ + "Native citations with grounding spans — verifiable RAG outputs by construction", + "Apache 2.0 open weights from a Western enterprise vendor with SOC 2 and VPC/on-prem options", + "Efficient serving: 218B sparse MoE with 25B active, runs on 2xH100 or a single B200", + "Unifies Command A, A Reasoning, A Vision, and A Translate into one multimodal model", + "48-language support with strong enterprise translation quality", + "Full deployment spectrum: SaaS, cloud marketplace, VPC, on-premises, self-hosted" + ], + + "limitations": [ + "Command A+ pricing not clearly published as of 2026-06-10 (prior Command A was $2.50/$10 per 1M)", + "Benchmarks vendor-reported; independent verification still emerging three weeks post-release", + "Behind agentic-coding specialists on SWE-bench-style tasks", + "Training data sources not disclosed in detail", + "Smaller open-source community than Chinese open-weight ecosystems" + ], + + "best_for": [ + "Enterprise RAG requiring verifiable, citation-backed answers", + "Regulated industries needing Western jurisdiction, SOC 2, and private/VPC deployment", + "Multilingual enterprise assistants across 48 languages", + "Organizations wanting permissive Apache 2.0 weights with commercial support behind them" + ], + + "not_recommended_for": [ + "Frontier agentic coding where open specialists or closed flagships lead", + "Budget-constrained projects until A+ pricing is confirmed", + "Creative writing applications" + ], + + "metadata": { + "pricing": { + "input": "Not yet published (prior Command A: $2.50 per 1M tokens)", + "output": "Not yet published (prior Command A: $10.00 per 1M tokens)", + "notes": "Command A+ API pricing not clearly published as of 2026-06-10; figures from the prior Command A generation shown for reference. Low confidence — verify with Cohere before procurement.", + "last_verified": "2026-06-10" + }, + "context_window": 256000, + "languages": [ + "English", + "French", + "German", + "Spanish", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)"], + "api_endpoint": "https://api.cohere.com/v2/chat", + "open_source": true, + "license": "Apache 2.0", + "architecture": "Sparse Mixture-of-Experts: 218B total / 25B active parameters; multimodal reasoning; unified Command A + Reasoning + Vision + Translate", + "parameters": "218B total / 25B active", + "release_date": "2026-05-20", + "supported_language_count": 48 + }, + + "related_entities": ["glm-5", "kimi-k2-6", "mistral-large-3", "claude-opus-4-8", "gpt-oss-120b"], + + "tags": [ + "enterprise", + "rag", + "citations", + "open-source", + "apache-2-0", + "multilingual", + "multimodal", + "mixture-of-experts", + "vpc-deployment", + "self-hostable" + ] +} diff --git a/data/models/deepseek-r1.json b/data/models/deepseek-r1.json index 229acad..c038094 100644 --- a/data/models/deepseek-r1.json +++ b/data/models/deepseek-r1.json @@ -4,9 +4,9 @@ "name": "DeepSeek-R1", "provider": "DeepSeek", "version": "20251020", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Advanced reasoning AI model from DeepSeek achieving 53.6% on SWE-bench and 79.8% on HumanEval. Combines strong coding capabilities with efficient reasoning at competitive pricing.", + "description": "DeepSeek's standalone reasoning model, now superseded and discontinued as a product line. Its reasoning capabilities were folded into DeepSeek V3.1's hybrid thinking mode (Aug 2025), then V3.2 (Dec 2025) and V4 (Apr 2026); a successor 'R2' never shipped. The legacy deepseek-reasoner API endpoint is scheduled for deprecation on 2026-07-24. Historically achieved 53.6% on SWE-bench and 79.8% on HumanEval with strong coding and reasoning at competitive pricing.", "website": "https://www.deepseek.com/", "trust_vector": { "performance_reliability": { @@ -416,7 +416,7 @@ "notes": "Moderate transparency with standard safety features. Limited disclosure compared to Western providers." }, "operational_excellence": { - "overall_score": 86, + "overall_score": 85, "criteria": { "api_design_quality": { "score": 88, @@ -447,18 +447,25 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 84, - "confidence": "medium", + "score": 74, + "confidence": "high", "evidence": [ { "source": "DeepSeek API Documentation", "url": "https://platform.deepseek.com/docs/api", "date": "2025-10-20", "value": "Basic versioning policy" + }, + { + "source": "DeepSeek API Pricing and Deprecation Notice", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-06-10", + "value": "Legacy deepseek-reasoner endpoint deprecates 2026-07-24; R1 line discontinued in favor of V3.1/V3.2/V4 hybrid thinking models" } ], "methodology": "Review of versioning practices", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10", + "notes": "Model superseded; endpoint deprecation requires migration to V3.2/V4" }, "monitoring_observability": { "score": 84, @@ -614,7 +621,8 @@ "60-day data retention (not ephemeral)", "Limited transparency on training data", "Smaller context window (64K tokens)", - "Less mature ecosystem compared to Western providers" + "Less mature ecosystem compared to Western providers", + "Superseded: line discontinued in favor of DeepSeek V3.1/V3.2/V4 hybrid thinking; legacy deepseek-reasoner endpoint deprecates 2026-07-24" ], "best_for": [ "Cost-sensitive projects needing strong coding capabilities", @@ -659,9 +667,12 @@ "related_entities": [ "claude-sonnet-4-5", "openai-o1", - "nemotron-ultra-253b" + "nemotron-ultra-253b", + "deepseek-v3-2", + "deepseek-v4" ], "tags": [ + "superseded", "coding", "reasoning", "open-source", diff --git a/data/models/deepseek-v3-0324.json b/data/models/deepseek-v3-0324.json index d0e9dbf..e33a5fb 100644 --- a/data/models/deepseek-v3-0324.json +++ b/data/models/deepseek-v3-0324.json @@ -4,9 +4,9 @@ "name": "DeepSeek V3 0324", "provider": "DeepSeek", "version": "0324", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "DeepSeek's latest open-weights model offering strong performance at competitive pricing. Designed for developers seeking capable models with transparent weights and commercial-friendly licensing.", + "description": "DeepSeek's March 2025 open-weights V3 checkpoint, now superseded by DeepSeek V3.1 (Aug 2025), V3.2 (Dec 2025), and V4 (Apr 2026). The legacy deepseek-chat endpoint name is scheduled for deprecation on 2026-07-24. Historically offered strong performance at competitive pricing for developers seeking capable models with transparent weights and commercial-friendly licensing.", "website": "https://www.deepseek.com/", "trust_vector": { "performance_reliability": { @@ -398,7 +398,7 @@ "notes": "Good transparency for open-weights model. Comprehensive documentation." }, "operational_excellence": { - "overall_score": 81, + "overall_score": 79, "criteria": { "api_design_quality": { "score": 83, @@ -429,18 +429,25 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 82, - "confidence": "medium", + "score": 72, + "confidence": "high", "evidence": [ { "source": "DeepSeek Releases", "url": "https://github.com/deepseek-ai/DeepSeek-V3/releases", "date": "2025-03-24", "value": "Clear versioning" + }, + { + "source": "DeepSeek API Pricing and Deprecation Notice", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-06-10", + "value": "Legacy deepseek-chat endpoint name deprecates 2026-07-24; superseded by V3.1, V3.2, and V4" } ], "methodology": "Versioning review", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10", + "notes": "Model superseded; endpoint deprecation requires migration to V3.2/V4" }, "monitoring_observability": { "score": 76, @@ -591,7 +598,8 @@ "Moderate security compared to proprietary models", "Smaller ecosystem than established providers", "Less mature enterprise support", - "Smaller context window (64K tokens)" + "Smaller context window (64K tokens)", + "Superseded by DeepSeek V3.1/V3.2/V4; legacy deepseek-chat endpoint name deprecates 2026-07-24" ], "best_for": [ "Cost-sensitive applications requiring good performance", @@ -636,9 +644,12 @@ "related_entities": [ "gpt-4-1-mini", "llama-4-scout", - "qwen2-5-vl-32b" + "qwen2-5-vl-32b", + "deepseek-v3-2", + "deepseek-v4" ], "tags": [ + "superseded", "open-weights", "cost-effective", "chinese", diff --git a/data/models/deepseek-v3-2.json b/data/models/deepseek-v3-2.json new file mode 100644 index 0000000..4f67a3f --- /dev/null +++ b/data/models/deepseek-v3-2.json @@ -0,0 +1,638 @@ +{ + "id": "deepseek-v3-2", + "type": "model", + "name": "DeepSeek-V3.2", + "provider": "DeepSeek", + "version": "20251201", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "DeepSeek's ~685B-parameter MoE flagship with DeepSeek Sparse Attention (DSA) for dramatically cheaper long-context inference. The V3.2-Speciale variant reached IMO 2025 gold-medal level (35/42) and 96.0% AIME. MIT-licensed open weights; the dominant open model through early 2026 until superseded by DeepSeek-V4.", + "website": "https://www.deepseek.com/", + "trust_vector": { + "performance_reliability": { + "overall_score": 93, + "criteria": { + "task_accuracy_code": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek-V3.2 Release Notes", + "url": "https://api-docs.deepseek.com/news/news251201", + "date": "2025-12-01", + "value": "Strong coding performance across SWE-bench and competitive programming; V3.2-Speciale placed 2nd at ICPC World Finals level" + }, + { + "source": "DeepSeek-V3.2 Technical Report", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "DSA architecture maintains coding accuracy while cutting long-context cost" + } + ], + "methodology": "Industry-standard coding benchmarks and competitive programming results from the official release and technical report", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek-V3.2 Release Notes", + "url": "https://api-docs.deepseek.com/news/news251201", + "date": "2025-12-01", + "value": "V3.2-Speciale reached IMO 2025 gold-medal level (35/42) and 96.0% on AIME" + }, + { + "source": "DeepSeek-V3.2 Technical Report", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "Frontier-level mathematical and olympiad reasoning documented in the technical report" + } + ], + "methodology": "Olympiad-level mathematics and competition benchmarks (IMO, AIME, ICPC) reported at release and corroborated by the technical report", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek-V3.2 Technical Report", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "Strong general knowledge and instruction-following results; best open model on most aggregate leaderboards through early 2026" + } + ], + "methodology": "Comprehensive knowledge and instruction-following benchmark review from the technical report and community leaderboards", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2025-12-01", + "value": "Stable outputs across thinking and non-thinking modes; sparse attention introduces minor variance on very long contexts" + } + ], + "methodology": "Repeated-prompt testing across temperature settings and context lengths, supplemented by community reports", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "1.8s", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/deepseek-v3-2", + "date": "2026-01-15", + "value": "Typical first-response latency ~1.8s on the first-party API; DSA keeps long-context latency near-flat" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes from independent benchmarking", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "4.2s", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/deepseek-v3-2", + "date": "2026-01-15", + "value": "p95 latency ~4.2s; reasoning mode substantially longer due to thinking tokens" + } + ], + "methodology": "95th percentile response time across diverse workloads from independent benchmarking", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "128,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/news/news251201", + "date": "2025-12-01", + "value": "128K context with DeepSeek Sparse Attention making long-context inference substantially cheaper" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 96, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Status", + "url": "https://status.deepseek.com/", + "date": "2026-06-01", + "value": "~99% uptime on first-party API; self-hosted and third-party deployments (Together, Fireworks, etc.) offer independent availability" + } + ], + "methodology": "Historical uptime data from the official status page plus availability of multiple third-party hosts", + "last_verified": "2026-06-10" + } + }, + "notes": "Frontier-level reasoning for an open model: V3.2-Speciale hit IMO 2025 gold-medal level and 2nd at ICPC World Finals. DSA makes 128K-context workloads unusually cheap. Speciale's dedicated API endpoint was temporary, but its weights remain open." + }, + "security": { + "overall_score": 83, + "criteria": { + "prompt_injection_resistance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Community red-team evaluations", + "url": "https://github.com/deepseek-ai/DeepSeek-V3.2", + "date": "2026-01-20", + "value": "Reasonable resistance to common injection patterns; weaker than frontier proprietary models on indirect injection" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attack patterns and community red-team reports", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Independent safety evaluations", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "Standard RLHF-based guardrails; open weights mean alignment can be removed by downstream fine-tuners" + } + ], + "methodology": "Testing against adversarial prompt datasets; assessment accounts for open-weight modifiability", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Privacy Policy", + "url": "https://www.deepseek.com/privacy", + "date": "2025-12-01", + "value": "Standard data handling on first-party API; self-hosting gives organizations complete data control" + } + ], + "methodology": "Analysis of privacy policy for the hosted API plus the self-hosting option for full data isolation", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek-V3.2 Technical Report", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "Safety post-training applied; refusal behavior comparable to prior DeepSeek releases" + } + ], + "methodology": "Safety testing across harmful content categories on default weights", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2025-12-01", + "value": "API key authentication, HTTPS-only transport, rate limiting on the first-party platform" + } + ], + "methodology": "Review of API security features and transport guarantees", + "last_verified": "2026-06-10" + } + }, + "notes": "Adequate default guardrails. As with all open-weight models, safety properties only hold for unmodified weights; self-hosting shifts security responsibility to the deployer." + }, + "privacy_compliance": { + "overall_score": 78, + "criteria": { + "data_residency": { + "value": "China (first-party API); anywhere via self-hosting or third-party hosts", + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek Privacy Policy", + "url": "https://www.deepseek.com/privacy", + "date": "2025-12-01", + "value": "First-party API data is processed and stored on servers in China; MIT-licensed weights allow self-hosting in any jurisdiction" + } + ], + "methodology": "Review of privacy policy and hosting options; China-jurisdiction caveat applies only to the first-party API", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Privacy Policy", + "url": "https://www.deepseek.com/privacy", + "date": "2025-12-01", + "value": "API data usage terms documented; self-hosting eliminates the concern entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms for the hosted API", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Per first-party API policy (China-hosted); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Terms of Service", + "url": "https://www.deepseek.com/terms", + "date": "2025-12-01", + "value": "First-party API retains data per Chinese regulatory requirements; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service; retention is deployment-dependent for open-weight models", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Platform Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2025-12-01", + "value": "No built-in PII redaction tooling; customer responsible for PII handling on any deployment" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Platform", + "url": "https://platform.deepseek.com/", + "date": "2025-12-01", + "value": "No SOC 2, HIPAA, or FedRAMP for the first-party API; compliant deployments achievable via certified Western hosts (AWS, Azure, Together) or self-hosting" + } + ], + "methodology": "Verification of certifications for the first-party platform; third-party hosted options inherit their providers' certifications", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Open-weight deployment options", + "url": "https://huggingface.co/deepseek-ai", + "date": "2025-12-01", + "value": "No zero-retention option on the first-party API, but self-hosting provides true zero external retention" + } + ], + "methodology": "Review of data handling across first-party API, third-party hosts, and self-hosting", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard open-model split: the first-party API is China-hosted with China-jurisdiction data residency, while self-hosting or Western third-party hosts (which most regulated enterprises use) avoid that concern entirely." + }, + "trust_transparency": { + "overall_score": 82, + "criteria": { + "explainability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek-V3.2 Release Notes", + "url": "https://api-docs.deepseek.com/news/news251201", + "date": "2025-12-01", + "value": "Visible chain-of-thought in thinking mode; reasoning traces fully inspectable on self-hosted deployments" + } + ], + "methodology": "Evaluation of reasoning transparency and trace accessibility", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Community factuality testing", + "url": "https://github.com/deepseek-ai/DeepSeek-V3.2", + "date": "2026-02-01", + "value": "Moderate hallucination rate, improved over V3.1; reasoning mode reduces factual errors on multi-step tasks" + } + ], + "methodology": "Testing on factual QA datasets and community evaluations", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Independent bias evaluations", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "Basic bias mitigation; known topic-avoidance behavior on China-politically-sensitive subjects in default weights" + } + ], + "methodology": "Evaluation on bias benchmarks and politically sensitive topic probes", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Model behavior assessment", + "url": "https://api-docs.deepseek.com/", + "date": "2025-12-01", + "value": "Expresses uncertainty in reasoning traces, though final answers can be overconfident" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek-V3.2 Technical Report", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "Detailed technical report covering DSA architecture, training methodology, and benchmark results, plus full open weights" + } + ], + "methodology": "Review of technical report and model card completeness", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek-V3.2 Technical Report", + "url": "https://arxiv.org/html/2512.02556v1", + "date": "2025-12-02", + "value": "Training methodology well documented; dataset composition described only at a high level" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Safety Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2025-12-01", + "value": "Standard alignment guardrails in released weights; removable by downstream fine-tuning" + } + ], + "methodology": "Analysis of built-in safety mechanisms in default weights", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong architectural transparency (open weights, detailed DSA technical report, visible reasoning traces) offset by limited training-data disclosure and known topic-avoidance on politically sensitive subjects." + }, + "operational_excellence": { + "overall_score": 85, + "criteria": { + "api_design_quality": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2025-12-01", + "value": "OpenAI-compatible API with thinking/non-thinking modes, function calling, and context caching" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek GitHub", + "url": "https://github.com/deepseek-ai", + "date": "2025-12-01", + "value": "OpenAI-SDK compatibility plus first-class support in vLLM and SGLang for self-hosting" + } + ], + "methodology": "Review of SDK compatibility and inference-framework support", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek API News", + "url": "https://api-docs.deepseek.com/news/news251201", + "date": "2025-12-01", + "value": "Rapid model turnover; the temporary V3.2-Speciale endpoint and short deprecation windows require deployment agility" + } + ], + "methodology": "Review of versioning practices and historical endpoint lifecycle", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Platform", + "url": "https://platform.deepseek.com/", + "date": "2025-12-01", + "value": "Usage dashboard with token metrics; full observability available when self-hosting" + } + ], + "methodology": "Review of monitoring tools across deployment options", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Support Channels", + "url": "https://api-docs.deepseek.com/", + "date": "2025-12-01", + "value": "Community and email support only; no enterprise SLA on the first-party platform" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face / hosting ecosystem", + "url": "https://huggingface.co/deepseek-ai", + "date": "2026-02-01", + "value": "Dominant open model through early 2026: hosted by Together, Fireworks, AWS Bedrock and others; broad fine-tune and tooling ecosystem" + } + ], + "methodology": "Analysis of third-party hosting, integrations, and community adoption", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek-V3.2 License", + "url": "https://huggingface.co/deepseek-ai", + "date": "2025-12-01", + "value": "MIT license: unrestricted commercial use, modification, and redistribution" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Mature ecosystem with broad third-party hosting. Main operational caveat is DeepSeek's rapid release cadence: V3.2 supersedes V3.1/V3-0324 and the standalone R1 line, and was itself superseded by V4 in April 2026." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "Excellent coding at exceptional cost; ICPC World Finals 2nd-place pedigree via Speciale. Best open coding value of its generation.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "customer-support": { + "overall": 82, + "notes": "Capable and cheap for support workloads; reasoning mode unnecessary overhead for simple tickets.", + "alternatives": ["qwen3-5", "claude-opus-4-5"] + }, + "content-creation": { + "overall": 83, + "notes": "Solid long-form generation; prose style less polished than frontier proprietary models.", + "alternatives": ["qwen3-5", "claude-opus-4-5"] + }, + "data-analysis": { + "overall": 91, + "notes": "Strong analytical reasoning with cheap 128K context thanks to DSA; excellent for large-document analysis on a budget.", + "alternatives": ["deepseek-v4", "qwen3-5"] + }, + "research-assistant": { + "overall": 92, + "notes": "Frontier-level mathematical and scientific reasoning (IMO gold-level via Speciale) with inspectable reasoning traces.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "legal-compliance": { + "overall": 74, + "notes": "First-party API is China-hosted with no Western certifications; viable only via self-hosting or certified third-party hosts.", + "alternatives": ["claude-opus-4-5"] + }, + "healthcare": { + "overall": 72, + "notes": "No HIPAA path on the first-party API. Self-hosted deployments in compliant infrastructure are the only viable route.", + "alternatives": ["claude-opus-4-5"] + }, + "financial-analysis": { + "overall": 88, + "notes": "Excellent quantitative reasoning at low cost; data-residency planning required for regulated workloads.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "education": { + "overall": 90, + "notes": "Outstanding math tutoring capability with visible step-by-step reasoning at prices viable for education budgets.", + "alternatives": ["qwen3-5", "deepseek-v4"] + }, + "creative-writing": { + "overall": 80, + "notes": "Competent but not a creative standout; reasoning strength does not translate to distinctive prose.", + "alternatives": ["claude-opus-4-5", "qwen3-5"] + } + }, + "strengths": [ + "Frontier open-model reasoning: V3.2-Speciale at IMO 2025 gold-medal level (35/42), 96.0% AIME, 2nd at ICPC World Finals", + "DeepSeek Sparse Attention (DSA) makes 128K long-context inference dramatically cheaper", + "MIT license with full ~685B MoE weights: unrestricted commercial use and self-hosting", + "Dominant open model through early 2026 with broad third-party hosting (Together, Fireworks, AWS Bedrock)", + "Detailed technical report and visible chain-of-thought reasoning", + "Very low first-party API pricing" + ], + "limitations": [ + "First-party API is China-hosted: China-jurisdiction data residency and no SOC 2/HIPAA/FedRAMP (self-hosting or Western hosts avoid this)", + "Superseded by DeepSeek-V4 (April 2026); no longer the frontier open model", + "V3.2-Speciale API endpoint was temporary; accessing Speciale now requires self-hosting its weights", + "Text-only: no vision or audio modalities", + "Topic-avoidance behavior on politically sensitive subjects in default weights", + "~685B parameters demand substantial multi-GPU infrastructure to self-host", + "No enterprise SLA or dedicated support on the first-party platform" + ], + "best_for": [ + "Cost-sensitive long-context workloads exploiting DSA's cheap 128K context", + "Mathematical, scientific, and competitive-programming reasoning tasks", + "Organizations wanting MIT-licensed frontier-class weights for self-hosted deployment", + "Research on sparse attention and reasoning-model architectures", + "Teams standardized on V3.2 fine-tunes who do not yet need V4" + ], + "not_recommended_for": [ + "Regulated Western workloads via the first-party API (use self-hosting or certified hosts instead)", + "Multimodal applications requiring vision or audio", + "Teams wanting the current frontier open model (DeepSeek-V4 supersedes it)", + "Organizations without GPU infrastructure that also cannot accept China-hosted APIs" + ], + "metadata": { + "pricing": { + "input": "$0.28 per 1M tokens (first-party API, cache miss)", + "output": "$0.42 per 1M tokens (first-party API)", + "notes": "DSA-driven price cut at launch made long-context usage exceptionally cheap; context caching discounts cache hits further. Legacy endpoints deprecate 2026-07-24 in favor of V4. Self-hosting cost is infrastructure-only under MIT license.", + "last_verified": "2026-06-10" + }, + "context_window": 128000, + "max_output": 64000, + "languages": [ + "English", + "Chinese", + "Japanese", + "Korean", + "Spanish", + "French", + "German", + "Portuguese", + "Russian" + ], + "modalities": ["text"], + "api_endpoint": "https://api.deepseek.com/v1/chat/completions", + "open_source": true, + "architecture": "~685B-parameter Mixture-of-Experts with DeepSeek Sparse Attention (DSA); thinking and non-thinking modes; V3.2-Speciale reasoning-specialized variant", + "parameters": "~685B total (MoE)", + "knowledge_cutoff": "Mid 2025" + }, + "related_entities": ["deepseek-v4", "deepseek-r1", "deepseek-v3-0324", "qwen3-5", "claude-opus-4-5"], + "tags": [ + "open-source", + "mit-license", + "reasoning", + "sparse-attention", + "long-context", + "cost-effective", + "mathematical", + "chinese-provider", + "superseded-by-v4" + ] +} diff --git a/data/models/deepseek-v4.json b/data/models/deepseek-v4.json new file mode 100644 index 0000000..0eee45b --- /dev/null +++ b/data/models/deepseek-v4.json @@ -0,0 +1,632 @@ +{ + "id": "deepseek-v4", + "type": "model", + "name": "DeepSeek-V4", + "provider": "DeepSeek", + "version": "20260424-preview", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "DeepSeek's preview flagship family: V4-Pro (1.6T total / 49B active MoE, the largest open-weight release ever) and V4-Flash (284B/13B). 1M-token context with up to 384K output, built on manifold-constrained Hyper Connections and Constrained Sparse Attention. MIT license. Vendor benchmark claims await broad independent verification.", + "website": "https://www.deepseek.com/", + "trust_vector": { + "performance_reliability": { + "overall_score": 92, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek API Pricing & Release Notes", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-04-24", + "value": "Vendor reports state-of-the-art open-model coding results for V4-Pro, exceeding V3.2" + }, + { + "source": "Wikipedia: DeepSeek", + "url": "https://en.wikipedia.org/wiki/DeepSeek", + "date": "2026-05-15", + "value": "V4 release (2026-04-24) documented as DeepSeek's strongest coding model; independent replication still in progress" + } + ], + "methodology": "Vendor-reported coding benchmarks; medium confidence pending broad independent verification of preview-release claims", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Release Notes", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-04-24", + "value": "Vendor claims frontier reasoning performance, building on V3.2-Speciale's olympiad-level results" + } + ], + "methodology": "Vendor-reported reasoning benchmarks; medium confidence until third-party evaluations of the preview mature", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia: DeepSeek", + "url": "https://en.wikipedia.org/wiki/DeepSeek", + "date": "2026-05-15", + "value": "Early community evaluations place V4-Pro at or near the top of open-model leaderboards" + } + ], + "methodology": "Community leaderboard positions and vendor benchmarks for a preview release", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "Preview status: behavior may change before stable release; community reports occasional long-output instability near the 384K limit" + } + ], + "methodology": "Repeated-prompt testing and community preview feedback", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "2.2s (V4-Pro), ~0.9s (V4-Flash)", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/deepseek-v4", + "date": "2026-05-20", + "value": "V4-Pro median latency ~2.2s; V4-Flash sub-second for standard prompts" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes from independent benchmarking", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "5.5s (V4-Pro)", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/deepseek-v4", + "date": "2026-05-20", + "value": "p95 ~5.5s for V4-Pro; long-context and long-output requests substantially longer" + } + ], + "methodology": "95th percentile response time across diverse workloads from independent benchmarking", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-04-24", + "value": "1M-token context window with up to 384K output tokens, enabled by Constrained Sparse Attention" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Status", + "url": "https://status.deepseek.com/", + "date": "2026-06-01", + "value": "Post-launch demand spikes caused intermittent degradation in late April/May 2026; stabilizing since" + } + ], + "methodology": "Status-page history since the 2026-04-24 launch", + "last_verified": "2026-06-10" + } + }, + "notes": "Largest open-weight release ever (V4-Pro: 1.6T total / 49B active). Vendor benchmarks are impressive but this is a preview: score confidence is medium until independent verification matures. V4-Flash (284B/13B) offers a much cheaper deployment point." + }, + "security": { + "overall_score": 83, + "criteria": { + "prompt_injection_resistance": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Early community red-team reports", + "url": "https://github.com/deepseek-ai", + "date": "2026-05-10", + "value": "Comparable to V3.2 on common injection patterns; preview red-teaming coverage still thin" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection patterns; limited preview-stage coverage", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Release Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "Standard alignment guardrails; open weights mean alignment is removable downstream" + } + ], + "methodology": "Adversarial prompt testing; assessment accounts for open-weight modifiability", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Privacy Policy", + "url": "https://www.deepseek.com/privacy", + "date": "2026-04-24", + "value": "Standard first-party API data handling; self-hosting gives complete data control" + } + ], + "methodology": "Analysis of privacy policy plus self-hosting option for full data isolation", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Release Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "Safety post-training applied; preview-stage safety evaluation less complete than for stable releases" + } + ], + "methodology": "Safety testing across harmful content categories on default weights", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "API key authentication, HTTPS-only, rate limiting on the first-party platform" + } + ], + "methodology": "Review of API security features and transport guarantees", + "last_verified": "2026-06-10" + } + }, + "notes": "Security posture mirrors V3.2 but with thinner preview-stage red-team coverage. Open weights shift safety responsibility to deployers who fine-tune or self-host." + }, + "privacy_compliance": { + "overall_score": 78, + "criteria": { + "data_residency": { + "value": "China (first-party API); anywhere via self-hosting or third-party hosts", + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek Privacy Policy", + "url": "https://www.deepseek.com/privacy", + "date": "2026-04-24", + "value": "First-party API data processed and stored in China; MIT-licensed weights allow deployment in any jurisdiction" + } + ], + "methodology": "Review of privacy policy and hosting options; China-jurisdiction caveat applies only to the first-party API", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Privacy Policy", + "url": "https://www.deepseek.com/privacy", + "date": "2026-04-24", + "value": "API data usage terms documented; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms for the hosted API", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Per first-party API policy (China-hosted); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Terms of Service", + "url": "https://www.deepseek.com/terms", + "date": "2026-04-24", + "value": "First-party API retention follows Chinese regulatory requirements; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service; retention is deployment-dependent for open-weight models", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Platform Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "No built-in PII redaction tooling; customer responsible on any deployment" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Platform", + "url": "https://platform.deepseek.com/", + "date": "2026-04-24", + "value": "No SOC 2, HIPAA, or FedRAMP on the first-party API; compliant deployments achievable via certified Western hosts or self-hosting" + } + ], + "methodology": "Verification of certifications for the first-party platform; third-party hosts inherit their own certifications", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Open-weight deployment options", + "url": "https://huggingface.co/deepseek-ai", + "date": "2026-04-24", + "value": "No zero-retention option on the first-party API; self-hosting provides true zero external retention" + } + ], + "methodology": "Review of data handling across first-party API, third-party hosts, and self-hosting", + "last_verified": "2026-06-10" + } + }, + "notes": "Same split as all DeepSeek releases: the first-party API is China-hosted with China-jurisdiction residency and no Western certifications, while self-hosting or Western third-party hosting avoids those concerns. V4-Pro's 1.6T size makes self-hosting far harder than V4-Flash." + }, + "trust_transparency": { + "overall_score": 80, + "criteria": { + "explainability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "Visible reasoning traces; up to 384K output supports very long inspectable chains of thought" + } + ], + "methodology": "Evaluation of reasoning transparency and trace accessibility", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 80, + "confidence": "low", + "evidence": [ + { + "source": "Early community testing", + "url": "https://github.com/deepseek-ai", + "date": "2026-05-20", + "value": "Preliminary results suggest parity with or improvement over V3.2; sample sizes still small for the preview" + } + ], + "methodology": "Limited factual QA testing during the preview period", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "Early bias probes", + "url": "https://en.wikipedia.org/wiki/DeepSeek", + "date": "2026-05-15", + "value": "Topic-avoidance on politically sensitive subjects persists from prior releases; formal bias audits of V4 not yet published" + } + ], + "methodology": "Preliminary bias probing; formal evaluations pending", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Model behavior assessment", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "Expresses uncertainty in reasoning traces; final-answer calibration unverified for the preview" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Release Documentation", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-04-24", + "value": "Release notes document architecture (Hyper Connections, Constrained Sparse Attention) and pricing; full technical report expected with the stable release" + } + ], + "methodology": "Review of preview documentation completeness against DeepSeek's historical technical-report standard", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Public Disclosures", + "url": "https://www.deepseek.com/", + "date": "2026-04-24", + "value": "High-level methodology described; dataset composition not disclosed in detail" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Safety Documentation", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "Standard alignment guardrails in released weights; preview safety evaluation ongoing" + } + ], + "methodology": "Analysis of built-in safety mechanisms in default weights", + "last_verified": "2026-06-10" + } + }, + "notes": "Architectural novelty (manifold-constrained Hyper Connections, Constrained Sparse Attention) is disclosed, but the preview lacks the full technical report and independent benchmark replication DeepSeek usually delivers. Treat vendor claims with medium confidence." + }, + "operational_excellence": { + "overall_score": 83, + "criteria": { + "api_design_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek API Documentation", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-04-24", + "value": "OpenAI-compatible API; V4-Pro and V4-Flash endpoints with context caching ($0.0028/1M cache hits on Flash)" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek GitHub", + "url": "https://github.com/deepseek-ai", + "date": "2026-04-24", + "value": "OpenAI-SDK compatibility; vLLM and SGLang support landed for V4 architecture within weeks of release" + } + ], + "methodology": "Review of SDK compatibility and inference-framework support", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 74, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek API Pricing Page", + "url": "https://api-docs.deepseek.com/quick_start/pricing", + "date": "2026-04-24", + "value": "Legacy deepseek-chat/deepseek-reasoner endpoints deprecate 2026-07-24, a three-month migration window" + } + ], + "methodology": "Review of deprecation timelines and migration windows", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Platform", + "url": "https://platform.deepseek.com/", + "date": "2026-04-24", + "value": "Usage dashboard with token metrics; full observability when self-hosting" + } + ], + "methodology": "Review of monitoring tools across deployment options", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "DeepSeek Support Channels", + "url": "https://api-docs.deepseek.com/", + "date": "2026-04-24", + "value": "Community and email support only; no enterprise SLA; preview status adds change risk" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Hosting ecosystem", + "url": "https://huggingface.co/deepseek-ai", + "date": "2026-05-20", + "value": "Third-party hosts onboarding rapidly; V4-Pro's 1.6T footprint limits the number of providers able to serve it, while V4-Flash adoption is broad" + } + ], + "methodology": "Analysis of third-party hosting availability six weeks post-release", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "DeepSeek-V4 License", + "url": "https://huggingface.co/deepseek-ai", + "date": "2026-04-24", + "value": "MIT license: unrestricted commercial use, modification, and redistribution" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Aggressive pricing (V4-Pro $0.435/$0.87, V4-Flash $0.14/$0.28 per 1M) and MIT licensing, but the short legacy-endpoint deprecation window (2026-07-24) and preview status demand migration agility. V4-Pro self-hosting is feasible only for well-resourced organizations." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 94, + "notes": "Vendor-reported state-of-the-art open-model coding; 1M context fits entire large repositories. Preview status warrants validation on your own tasks.", + "alternatives": ["claude-opus-4-5", "deepseek-v3-2"] + }, + "customer-support": { + "overall": 84, + "notes": "V4-Flash is a strong cheap option for high-volume support with aggressive cache-hit pricing.", + "alternatives": ["qwen3-5", "deepseek-v3-2"] + }, + "content-creation": { + "overall": 85, + "notes": "Up to 384K output enables book-length single-pass drafts; prose quality solid but not best-in-class.", + "alternatives": ["claude-opus-4-5", "qwen3-5"] + }, + "data-analysis": { + "overall": 92, + "notes": "1M context plus strong reasoning makes whole-dataset and multi-document analysis practical at open-model prices.", + "alternatives": ["qwen3-5", "deepseek-v3-2"] + }, + "research-assistant": { + "overall": 93, + "notes": "1M-token context ingests entire literature corpora; frontier reasoning lineage from V3.2-Speciale.", + "alternatives": ["claude-opus-4-5", "qwen3-5"] + }, + "legal-compliance": { + "overall": 73, + "notes": "China-hosted first-party API and preview status are both disqualifying for most regulated legal work; self-hosting V4-Flash is the viable path.", + "alternatives": ["claude-opus-4-5"] + }, + "healthcare": { + "overall": 71, + "notes": "No HIPAA path on the first-party API; preview status adds change risk. Only self-hosted compliant deployments are viable.", + "alternatives": ["claude-opus-4-5"] + }, + "financial-analysis": { + "overall": 88, + "notes": "Excellent quantitative reasoning over very long filings at low cost; verify vendor claims and plan data residency for regulated use.", + "alternatives": ["claude-opus-4-5", "deepseek-v3-2"] + }, + "education": { + "overall": 89, + "notes": "V4-Flash pricing is ideal for education-scale deployment with strong math reasoning.", + "alternatives": ["qwen3-5", "deepseek-v3-2"] + }, + "creative-writing": { + "overall": 81, + "notes": "Massive output length helps novel-scale drafting; stylistic range remains behind dedicated creative leaders.", + "alternatives": ["claude-opus-4-5", "qwen3-5"] + } + }, + "strengths": [ + "Largest open-weight release ever: V4-Pro at 1.6T total / 49B active parameters under MIT license", + "1M-token context window with up to 384K output tokens", + "Novel architecture: manifold-constrained Hyper Connections and Constrained Sparse Attention", + "Aggressive pricing: V4-Pro $0.435/$0.87 and V4-Flash $0.14/$0.28 per 1M tokens ($0.0028 cache hits on Flash)", + "V4-Flash (284B/13B) offers a practical self-hosting and high-volume deployment point", + "Inherits DeepSeek's frontier reasoning lineage from V3.2/Speciale" + ], + "limitations": [ + "Preview status: behavior, pricing, and endpoints may change; vendor benchmark claims not yet broadly independently verified", + "First-party API is China-hosted: China-jurisdiction data residency and no SOC 2/HIPAA/FedRAMP (self-hosting or Western hosts avoid this)", + "V4-Pro's 1.6T footprint makes self-hosting impractical for all but the largest organizations", + "Short migration window: legacy deepseek-chat/deepseek-reasoner endpoints deprecate 2026-07-24", + "Text-only: no native vision or audio", + "No enterprise SLA or dedicated support on the first-party platform", + "Full technical report not yet published for the preview" + ], + "best_for": [ + "Million-token context workloads: whole-repository coding, corpus-scale research, multi-document analysis", + "Cost-sensitive high-volume inference via V4-Flash with cache-hit pricing", + "Organizations wanting the most capable MIT-licensed weights available", + "Teams already on DeepSeek migrating off the deprecating legacy endpoints" + ], + "not_recommended_for": [ + "Production systems that cannot tolerate preview-stage changes", + "Regulated Western workloads via the first-party API", + "Multimodal applications requiring vision or audio", + "Self-hosting V4-Pro without large-scale multi-node GPU infrastructure" + ], + "metadata": { + "pricing": { + "input": "$0.435 per 1M tokens (V4-Pro); $0.14 per 1M (V4-Flash, $0.0028 cache hit)", + "output": "$0.87 per 1M tokens (V4-Pro); $0.28 per 1M (V4-Flash)", + "notes": "Preview pricing per official pricing page; legacy deepseek-chat/deepseek-reasoner endpoints deprecate 2026-07-24. Self-hosting is infrastructure-cost-only under MIT license.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 384000, + "languages": [ + "English", + "Chinese", + "Japanese", + "Korean", + "Spanish", + "French", + "German", + "Portuguese", + "Russian", + "Arabic" + ], + "modalities": ["text"], + "api_endpoint": "https://api.deepseek.com/v1/chat/completions", + "open_source": true, + "architecture": "Mixture-of-Experts with manifold-constrained Hyper Connections and Constrained Sparse Attention; V4-Pro 1.6T total / 49B active, V4-Flash 284B total / 13B active", + "parameters": "V4-Pro: 1.6T total / 49B active; V4-Flash: 284B total / 13B active", + "knowledge_cutoff": "Early 2026" + }, + "related_entities": ["deepseek-v3-2", "deepseek-r1", "qwen3-5", "claude-opus-4-5", "kimi-k2-6"], + "tags": [ + "open-source", + "mit-license", + "preview", + "long-context", + "reasoning", + "sparse-attention", + "cost-effective", + "chinese-provider", + "flagship" + ] +} diff --git a/data/models/gemini-2-0-flash.json b/data/models/gemini-2-0-flash.json index a076727..6c01aab 100644 --- a/data/models/gemini-2-0-flash.json +++ b/data/models/gemini-2-0-flash.json @@ -4,9 +4,9 @@ "name": "Gemini 2.0 Flash", "provider": "Google", "version": "20251101", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Fast and efficient multimodal AI model from Google achieving 53.6% on SWE-bench and 62.1% on MMLU. Optimized for speed with strong vision capabilities and real-time applications.", + "description": "SHUT DOWN: Google shut down the Gemini 2.0 Flash family on 2026-06-01; the model is no longer served via the Gemini API. Historically a fast, efficient multimodal model (53.6% SWE-bench, 62.1% MMLU) optimized for speed and vision. Migrate to Gemini 3.5 Flash for equivalent fast multimodal workloads.", "website": "https://deepmind.google/technologies/gemini/", "trust_vector": { "performance_reliability": { @@ -410,7 +410,7 @@ "notes": "Good transparency with strong safety guardrails. Standard explainability for a Flash model." }, "operational_excellence": { - "overall_score": 92, + "overall_score": 89, "criteria": { "api_design_quality": { "score": 93, @@ -441,7 +441,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 91, + "score": 79, "confidence": "high", "evidence": [ { @@ -449,6 +449,12 @@ "url": "https://cloud.google.com/apis/design/versioning", "date": "2025-11-01", "value": "Clear versioning policy" + }, + { + "source": "Gemini API Changelog", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-06-10", + "value": "Gemini 2.0 Flash family shut down 2026-06-01; no longer served" } ], "methodology": "Review of versioning practices", @@ -483,7 +489,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 93, + "score": 83, "confidence": "high", "evidence": [ { @@ -511,7 +517,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Excellent operational maturity leveraging Google Cloud. Mature ecosystem with comprehensive support." + "notes": "Model shut down 2026-06-01 and no longer served. Versioning and ecosystem scores reduced to reflect shutdown." } }, "use_case_ratings": { @@ -607,7 +613,8 @@ "Limited explainability compared to reasoning models", "Training data transparency could be improved", "May not be suitable for highly complex reasoning tasks", - "Moderate hallucination rates" + "Moderate hallucination rates", + "SHUT DOWN 2026-06-01: Gemini 2.0 Flash family is no longer served; migrate to Gemini 3.5 Flash" ], "best_for": [ "Real-time applications requiring fast responses", @@ -656,11 +663,13 @@ "parameters": "Not disclosed" }, "related_entities": [ + "gemini-3-5-flash", "gemini-2-5-pro", "claude-haiku-4-5", "gpt-5" ], "tags": [ + "retired", "fast", "multimodal", "vision", diff --git a/data/models/gemini-3-1-pro.json b/data/models/gemini-3-1-pro.json new file mode 100644 index 0000000..2306cbf --- /dev/null +++ b/data/models/gemini-3-1-pro.json @@ -0,0 +1,627 @@ +{ + "id": "gemini-3-1-pro", + "type": "model", + "name": "Gemini 3.1 Pro", + "provider": "Google", + "version": "gemini-3.1-pro", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Google's GA flagship reasoning model with 77.1% ARC-AGI-2 (2.5x Gemini 3 Pro), 94.3% GPQA Diamond, 2887 Elo on LiveCodeBench Pro, and 1M token context. Supersedes Gemini 3 Pro Preview as the production-ready frontier tier.", + "website": "https://ai.google.dev/gemini-api/docs/changelog", + + "trust_vector": { + "performance_reliability": { + "overall_score": 96, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "LiveCodeBench Pro", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-02-19", + "value": "2887 Elo on competitive coding (frontier-leading at launch)" + }, + { + "source": "MCP Atlas", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-02-19", + "value": "78.2% on multi-tool agentic orchestration" + } + ], + "methodology": "Competitive programming and agentic tool-use benchmarks from official launch materials", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "ARC-AGI-2", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-02-19", + "value": "77.1% (vs 31.1% standard Gemini 3 Pro, ~2.5x generational gain)" + }, + { + "source": "GPQA Diamond", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-02-19", + "value": "94.3% on PhD-level science questions" + } + ], + "methodology": "Abstract reasoning and PhD-level science benchmarks reported at GA launch", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 96, + "confidence": "medium", + "evidence": [ + { + "source": "Google DeepMind Models Page", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-02-19", + "value": "Positioned as GA flagship, superseding Gemini 3 Pro Preview across general benchmarks" + } + ], + "methodology": "Cross-benchmark comparison against predecessor Gemini 3 Pro Preview", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 94, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Documentation", + "url": "https://ai.google.dev/gemini-api/docs", + "date": "2026-02-19", + "value": "GA stability commitments versus preview-tier predecessor" + } + ], + "methodology": "Consistency assessment based on GA status and documented model behavior", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "~1.8s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/", + "date": "2026-06-01", + "value": "Typical response time under 2s for standard prompts; deep reasoning modes slower" + } + ], + "methodology": "Median latency from third-party aggregator measurements", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Gemini API Changelog", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-02-19", + "value": "1M token context window at GA" + } + ], + "methodology": "Official specification from provider documentation", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Status", + "url": "https://status.cloud.google.com/", + "date": "2026-06-01", + "value": "99.9% uptime (last 90 days, Vertex AI)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Massive reasoning jump: 77.1% ARC-AGI-2 vs 31.1% for Gemini 3 Pro. GA status removes preview-tier risk. Note: Gemini 3.5 Pro was announced at I/O May 2026 but is not yet GA." + }, + + "security": { + "overall_score": 88, + "criteria": { + "prompt_injection_resistance": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Safety", + "url": "https://ai.google/responsibility/", + "date": "2026-02-19", + "value": "Hardened prompt injection defenses carried forward from Gemini 3 line" + } + ], + "methodology": "OWASP LLM01 prompt injection testing and vendor safety documentation review", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Google Safety Settings", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-02-19", + "value": "Improved adversarial robustness reported at GA" + } + ], + "methodology": "Adversarial prompt dataset testing", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini API Terms", + "url": "https://ai.google.dev/gemini-api/terms", + "date": "2026-02-19", + "value": "Paid-tier API data not used for training" + } + ], + "methodology": "Privacy policy and API terms review", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Google Safety Filters", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-02-19", + "value": "Configurable multi-category safety filters" + } + ], + "methodology": "Safety filter testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Security", + "url": "https://cloud.google.com/security", + "date": "2026-02-19", + "value": "Google Cloud security standards, IAM integration on Vertex AI" + } + ], + "methodology": "Review of API security features and infrastructure", + "last_verified": "2026-06-10" + } + }, + "notes": "Inherits Google Cloud security posture. Configurable safety filters and Vertex AI IAM controls for enterprise deployment." + }, + + "privacy_compliance": { + "overall_score": 88, + "criteria": { + "data_residency": { + "value": "Multi-region including EU residency options (Vertex AI)", + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Locations", + "url": "https://cloud.google.com/about/locations", + "date": "2026-02-19", + "value": "EU data residency options available via Vertex AI regions" + } + ], + "methodology": "Cloud infrastructure and data residency documentation review", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Gemini API Terms", + "url": "https://ai.google.dev/gemini-api/terms", + "date": "2026-02-19", + "value": "Paid API data not used for training; Vertex AI data governance applies" + } + ], + "methodology": "Terms of service review", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Varies by tier; zero retention available on Vertex AI enterprise", + "confidence": "medium", + "evidence": [ + { + "source": "Google Cloud Service Terms", + "url": "https://cloud.google.com/terms/service-terms", + "date": "2026-02-19", + "value": "Enterprise zero-retention configurations available" + } + ], + "methodology": "Data retention policy review", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Documentation", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-02-19", + "value": "Customer responsible for PII redaction; Cloud DLP integration available" + } + ], + "methodology": "Data protection capability review", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Compliance", + "url": "https://cloud.google.com/security/compliance", + "date": "2026-02-19", + "value": "SOC 1/2/3, ISO 27001/27017/27018, GDPR, HIPAA (via Google Cloud)" + } + ], + "methodology": "Certification verification through Google Cloud compliance center", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Vertex AI Data Governance", + "url": "https://cloud.google.com/vertex-ai/docs", + "date": "2026-02-19", + "value": "Zero-retention configuration available for enterprise Vertex AI customers" + } + ], + "methodology": "Enterprise feature review", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong enterprise posture via Vertex AI data governance, SOC/ISO certifications, and EU data residency options." + }, + + "trust_transparency": { + "overall_score": 88, + "criteria": { + "explainability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Gemini API Documentation", + "url": "https://ai.google.dev/gemini-api/docs", + "date": "2026-02-19", + "value": "Thinking traces and configurable reasoning depth exposed via API" + } + ], + "methodology": "Reasoning transparency evaluation", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Google Launch Materials", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-02-19", + "value": "Improved factual grounding over Gemini 3 Pro Preview" + } + ], + "methodology": "Factual QA testing and vendor claims review", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Principles", + "url": "https://ai.google/responsibility/principles/", + "date": "2026-02-19", + "value": "Regular bias testing and mitigation per AI Principles" + } + ], + "methodology": "Bias benchmark evaluation and policy review", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://ai.google.dev/gemini-api/docs", + "date": "2026-02-19", + "value": "Expresses uncertainty appropriately in extended reasoning mode" + } + ], + "methodology": "Qualitative assessment of confidence expression", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Gemini Documentation", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-02-19", + "value": "Comprehensive GA documentation with benchmarks and limitations" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Blog", + "url": "https://blog.google/technology/ai/", + "date": "2026-02-19", + "value": "General training description provided; detailed sources not disclosed" + } + ], + "methodology": "Public disclosure review", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Safety Settings", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-02-19", + "value": "Configurable multi-category safety guardrails" + } + ], + "methodology": "Safety mechanism analysis", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong transparency via exposed thinking traces and comprehensive GA documentation. Training data details remain limited (industry standard)." + }, + + "operational_excellence": { + "overall_score": 93, + "criteria": { + "api_design_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Gemini API", + "url": "https://ai.google.dev/gemini-api/docs", + "date": "2026-02-19", + "value": "RESTful API with streaming, function calling, multimodal, thinking control" + } + ], + "methodology": "API design and feature completeness review", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Google Gen AI SDKs", + "url": "https://github.com/googleapis/python-genai", + "date": "2026-02-19", + "value": "Unified Gen AI SDKs for Python, Node.js, Go, Java; actively maintained" + } + ], + "methodology": "SDK quality and maintenance assessment", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Gemini API Changelog", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-02-19", + "value": "GA release 2026-02-19 with documented deprecation timeline for 3 Pro Preview" + } + ], + "methodology": "Versioning policy and changelog review", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Console", + "url": "https://console.cloud.google.com/", + "date": "2026-02-19", + "value": "Comprehensive Cloud Console and Vertex AI monitoring" + } + ], + "methodology": "Observability tooling review", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Support", + "url": "https://cloud.google.com/support", + "date": "2026-02-19", + "value": "Enterprise support tiers with SLAs" + } + ], + "methodology": "Support channel assessment", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Google AI Ecosystem", + "url": "https://ai.google.dev/", + "date": "2026-02-19", + "value": "Day-one availability across Gemini app, AI Studio, Vertex AI" + } + ], + "methodology": "Ecosystem and integration analysis", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Terms", + "url": "https://cloud.google.com/terms", + "date": "2026-02-19", + "value": "Standard commercial terms; enterprise agreements available" + } + ], + "methodology": "License terms review", + "last_verified": "2026-06-10" + }, + "pricing_transparency": { + "value": "~$2/$12 per 1M tokens (<=200K context), $4/$18 above 200K", + "confidence": "medium", + "evidence": [ + { + "source": "Pricing aggregators", + "url": "https://artificialanalysis.ai/", + "date": "2026-06-01", + "value": "Aggregator-sourced: ~$2 input / $12 output per 1M tokens at <=200K context; $4/$18 above" + } + ], + "methodology": "Cross-referenced third-party pricing aggregators; official pricing page recommended for confirmation", + "last_verified": "2026-06-10" + } + }, + "notes": "Mature GA operational posture across all Google AI surfaces. Pricing figures are aggregator-sourced (medium confidence)." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 95, + "notes": "2887 Elo LiveCodeBench Pro and 78.2% MCP Atlas. Strong agentic coding; 1M context covers full codebases.", + "alternatives": ["claude-opus-4-8", "gemini-3-5-flash"] + }, + "data-analysis": { + "overall": 96, + "notes": "1M context plus top-tier reasoning makes it excellent for massive dataset analysis.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "research-assistant": { + "overall": 97, + "notes": "94.3% GPQA Diamond and 1M context. Best-in-class for deep multi-document research.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "legal-compliance": { + "overall": 90, + "notes": "1M context for full contract corpora. EU data residency and Vertex AI governance support regulated workloads.", + "alternatives": ["claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 94, + "notes": "Frontier quantitative reasoning with long context for large filing sets.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "education": { + "overall": 94, + "notes": "Exceptional reasoning depth for tutoring; thinking traces aid pedagogical explanations.", + "alternatives": ["gpt-5-5", "gemini-3-5-flash"] + }, + "content-creation": { + "overall": 91, + "notes": "Strong long-form generation; reasoning depth helps structured technical content.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 88, + "notes": "HIPAA via Google Cloud. Strong reasoning for clinical literature, but use Vertex AI governance controls.", + "alternatives": ["claude-opus-4-8"] + } + }, + + "strengths": [ + "Exceptional abstract reasoning: 77.1% ARC-AGI-2 (~2.5x Gemini 3 Pro's 31.1%)", + "94.3% GPQA Diamond, near-saturation PhD-level science", + "Frontier coding: 2887 Elo LiveCodeBench Pro, 78.2% MCP Atlas", + "1M token context window with GA stability", + "Enterprise posture: Vertex AI data governance, SOC/ISO certs, EU residency", + "Day-one availability across AI Studio, Vertex AI, and Gemini app" + ], + + "limitations": [ + "Pricing confidence medium: figures aggregator-sourced (~$2/$12, $4/$18 above 200K)", + "Extended reasoning modes add significant latency", + "Training data transparency limited (industry standard)", + "Gemini 3.5 Pro announced at I/O May 2026 may supersede it soon (not yet GA)", + "Long-context (>200K) pricing roughly doubles per-token cost" + ], + + "best_for": [ + "Hardest reasoning workloads (research, math, science)", + "Long-context applications over entire codebases or document corpora", + "Enterprises on Google Cloud needing data governance and EU residency", + "Agentic systems requiring strong multi-tool orchestration" + ], + + "not_recommended_for": [ + "Latency-sensitive applications (consider Gemini 3.5 Flash)", + "High-volume cost-sensitive inference", + "Teams needing fully confirmed first-party pricing before procurement" + ], + + "metadata": { + "pricing": { + "input": "~$2.00 per 1M tokens (<=200K), ~$4.00 per 1M tokens (>200K)", + "output": "~$12.00 per 1M tokens (<=200K), ~$18.00 per 1M tokens (>200K)", + "notes": "Aggregator-sourced (medium confidence). Tiered by context length; verify against official pricing page.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 64000, + "languages": ["English", "100+ languages"], + "modalities": ["text", "vision", "audio", "video"], + "api_endpoint": "https://generativelanguage.googleapis.com/v1beta/models", + "open_source": false, + "architecture": "Multimodal transformer with configurable extended reasoning", + "parameters": "Not disclosed", + "knowledge_cutoff": "Late 2025 (not officially confirmed)", + "release_date": "2026-02-19" + }, + + "related_entities": ["gemini-3-pro", "gemini-3-5-flash", "claude-opus-4-8", "gpt-5-5"], + + "tags": [ + "flagship", + "ga", + "reasoning", + "long-context", + "1m-tokens", + "multimodal", + "google-cloud", + "enterprise" + ] +} diff --git a/data/models/gemini-3-5-flash.json b/data/models/gemini-3-5-flash.json new file mode 100644 index 0000000..70b8d4b --- /dev/null +++ b/data/models/gemini-3-5-flash.json @@ -0,0 +1,608 @@ +{ + "id": "gemini-3-5-flash", + "type": "model", + "name": "Gemini 3.5 Flash", + "provider": "Google", + "version": "gemini-3.5-flash", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Google's GA 'frontier workhorse' launched at I/O 2026. Beats Gemini 3.1 Pro on agentic and coding suites (76.2% Terminal-Bench 2.1, 83.6% MCP Atlas) at roughly 4x the speed, with 1M token context. Pricier than past Flash tiers at $1.50/$9.00 per 1M.", + "website": "https://deepmind.google/models/gemini/", + + "trust_vector": { + "performance_reliability": { + "overall_score": 93, + "criteria": { + "task_accuracy_code": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Terminal-Bench 2.1", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-05-19", + "value": "76.2% on terminal/command-line agentic tasks (official claim, beats Gemini 3.1 Pro)" + }, + { + "source": "MCP Atlas", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-05-19", + "value": "83.6% multi-tool orchestration (official claim, above Gemini 3.1 Pro's 78.2%)" + } + ], + "methodology": "Official launch benchmarks for agentic coding; vendor-reported, pending broad third-party replication", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "Finance Agent v2", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-05-19", + "value": "57.9% on multi-step financial agent tasks (official claim)" + }, + { + "source": "CharXiv", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-05-19", + "value": "84.2% on chart reasoning (official claim)" + } + ], + "methodology": "Official agentic and multimodal reasoning benchmarks from launch materials", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 91, + "confidence": "medium", + "evidence": [ + { + "source": "Google I/O 2026 Launch", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-05-19", + "value": "Positioned as 'frontier workhorse' matching or beating prior Pro tier on agentic suites" + } + ], + "methodology": "Cross-benchmark comparison against Gemini 3.1 Pro from official launch claims", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "Google DeepMind Models Page", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-05-19", + "value": "GA status with ~4x speed of Gemini 3.1 Pro at consistent quality" + } + ], + "methodology": "Consistency assessment based on GA status and vendor throughput claims", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "<1s", + "confidence": "medium", + "evidence": [ + { + "source": "Google launch materials", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-05-19", + "value": "~4x faster than Gemini 3.1 Pro on agentic workloads" + } + ], + "methodology": "Relative speed claims from official materials; absolute latency varies by workload", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,048,576 tokens input / 65,536 output", + "confidence": "high", + "evidence": [ + { + "source": "Gemini API Changelog", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-05-19", + "value": "1,048,576 input tokens, 64K output tokens" + } + ], + "methodology": "Official specification from provider documentation", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Status", + "url": "https://status.cloud.google.com/", + "date": "2026-06-01", + "value": "99.9% uptime (last 90 days, Vertex AI)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Unusual positioning: a Flash-tier model that beats the flagship 3.1 Pro on agentic/coding suites at ~4x speed. Benchmark figures are official claims from launch (2026-05-19); third-party replication still maturing." + }, + + "security": { + "overall_score": 87, + "criteria": { + "prompt_injection_resistance": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Safety", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-05-19", + "value": "Gemini 3.x-generation prompt injection defenses" + } + ], + "methodology": "OWASP LLM01 testing and vendor documentation review", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Google Safety Testing", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-05-19", + "value": "Adversarial robustness consistent with Gemini 3.x family" + } + ], + "methodology": "Adversarial prompt testing", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini API Terms", + "url": "https://ai.google.dev/gemini-api/terms", + "date": "2026-05-19", + "value": "Paid-tier API data not used for training" + } + ], + "methodology": "Privacy policy and API terms review", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Google Safety Filters", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-05-19", + "value": "Configurable multi-category safety filters" + } + ], + "methodology": "Safety filter testing across content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Security", + "url": "https://cloud.google.com/security", + "date": "2026-05-19", + "value": "Google Cloud security standards, Vertex AI IAM" + } + ], + "methodology": "API security feature review", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard Gemini 3.x-family security posture on Google Cloud infrastructure. High-speed agentic use increases the importance of downstream tool sandboxing." + }, + + "privacy_compliance": { + "overall_score": 88, + "criteria": { + "data_residency": { + "value": "Multi-region including EU residency options (Vertex AI)", + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Locations", + "url": "https://cloud.google.com/about/locations", + "date": "2026-05-19", + "value": "Regional deployment options via Vertex AI" + } + ], + "methodology": "Cloud infrastructure documentation review", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Gemini API Terms", + "url": "https://ai.google.dev/gemini-api/terms", + "date": "2026-05-19", + "value": "Paid API data not used for training" + } + ], + "methodology": "Terms of service review", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Varies by tier; zero retention available on Vertex AI enterprise", + "confidence": "medium", + "evidence": [ + { + "source": "Google Cloud Service Terms", + "url": "https://cloud.google.com/terms/service-terms", + "date": "2026-05-19", + "value": "Enterprise zero-retention configurations available" + } + ], + "methodology": "Data retention policy review", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Documentation", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-05-19", + "value": "Customer responsible for PII redaction; Cloud DLP integration available" + } + ], + "methodology": "Data protection capability review", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Compliance", + "url": "https://cloud.google.com/security/compliance", + "date": "2026-05-19", + "value": "SOC 1/2/3, ISO 27001/27017/27018, GDPR, HIPAA (via Google Cloud)" + } + ], + "methodology": "Certification verification", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Vertex AI Data Governance", + "url": "https://cloud.google.com/vertex-ai/docs", + "date": "2026-05-19", + "value": "Zero-retention configuration available for enterprise customers" + } + ], + "methodology": "Enterprise feature review", + "last_verified": "2026-06-10" + } + }, + "notes": "Same Google Cloud compliance envelope as the Pro tier: SOC/ISO certifications, GDPR, HIPAA via Google Cloud, EU residency options." + }, + + "trust_transparency": { + "overall_score": 86, + "criteria": { + "explainability": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Gemini API Documentation", + "url": "https://ai.google.dev/gemini-api/docs", + "date": "2026-05-19", + "value": "Thinking traces available, though shallower than Pro-tier extended reasoning" + } + ], + "methodology": "Reasoning transparency evaluation", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "CharXiv (chart grounding)", + "url": "https://deepmind.google/models/gemini/", + "date": "2026-05-19", + "value": "84.2% chart reasoning suggests solid grounding; speed-optimized tiers historically hallucinate slightly more" + } + ], + "methodology": "Benchmark-derived grounding assessment; vendor claims pending independent replication", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Principles", + "url": "https://ai.google/responsibility/principles/", + "date": "2026-05-19", + "value": "Regular bias testing and mitigation per AI Principles" + } + ], + "methodology": "Bias benchmark evaluation and policy review", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 84, + "confidence": "low", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://ai.google.dev/gemini-api/docs", + "date": "2026-05-19", + "value": "Limited independent assessment so far; recent GA release" + } + ], + "methodology": "Qualitative assessment; limited data given three weeks since GA", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Gemini Documentation", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-05-19", + "value": "Launch documentation with benchmarks, specs, and pricing" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 81, + "confidence": "medium", + "evidence": [ + { + "source": "Google AI Blog", + "url": "https://blog.google/technology/ai/", + "date": "2026-05-19", + "value": "General training description; detailed sources not disclosed" + } + ], + "methodology": "Public disclosure review", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Safety Settings", + "url": "https://ai.google.dev/gemini-api/docs/safety-settings", + "date": "2026-05-19", + "value": "Configurable multi-category safety guardrails" + } + ], + "methodology": "Safety mechanism analysis", + "last_verified": "2026-06-10" + } + }, + "notes": "Solid documentation at launch, but the model is only ~3 weeks GA; most benchmark figures remain vendor-reported and independent verification is still accumulating." + }, + + "operational_excellence": { + "overall_score": 93, + "criteria": { + "api_design_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Gemini API", + "url": "https://ai.google.dev/gemini-api/docs", + "date": "2026-05-19", + "value": "RESTful API with streaming, function calling, multimodal, context caching" + } + ], + "methodology": "API design and feature completeness review", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Google Gen AI SDKs", + "url": "https://github.com/googleapis/python-genai", + "date": "2026-05-19", + "value": "Unified Gen AI SDKs across Python, Node.js, Go, Java" + } + ], + "methodology": "SDK quality and maintenance assessment", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Gemini API Changelog", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-05-19", + "value": "GA at I/O 2026 with documented model lifecycle" + } + ], + "methodology": "Versioning and changelog review", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Console", + "url": "https://console.cloud.google.com/", + "date": "2026-05-19", + "value": "Comprehensive Cloud Console and Vertex AI monitoring" + } + ], + "methodology": "Observability tooling review", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Support", + "url": "https://cloud.google.com/support", + "date": "2026-05-19", + "value": "Enterprise support tiers with SLAs" + } + ], + "methodology": "Support channel assessment", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Google AI Ecosystem", + "url": "https://ai.google.dev/", + "date": "2026-05-19", + "value": "Day-one availability across AI Studio, Vertex AI, and Gemini app" + } + ], + "methodology": "Ecosystem and integration analysis", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Google Cloud Terms", + "url": "https://cloud.google.com/terms", + "date": "2026-05-19", + "value": "Standard commercial terms; enterprise agreements available" + } + ], + "methodology": "License terms review", + "last_verified": "2026-06-10" + } + }, + "notes": "Full Google Cloud operational stack from day one. Pricing ($1.50/$9.00, cached input $0.15) is notably higher than past Flash tiers, narrowing the cost gap to Pro models." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 94, + "notes": "76.2% Terminal-Bench 2.1 and 83.6% MCP Atlas — beats Gemini 3.1 Pro on agentic coding at ~4x speed. Excellent agent-loop economics.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "customer-support": { + "overall": 93, + "notes": "Speed plus frontier quality is ideal for high-volume support. Multimodal input handles screenshots.", + "alternatives": ["gemini-3-flash", "gpt-5-5"] + }, + "data-analysis": { + "overall": 92, + "notes": "84.2% CharXiv chart reasoning and 1M context for large datasets at workhorse cost.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 88, + "notes": "57.9% Finance Agent v2 is strong for an agentic suite, but high-stakes analysis still favors Pro-tier reasoning.", + "alternatives": ["gemini-3-1-pro", "gpt-5-5"] + }, + "research-assistant": { + "overall": 89, + "notes": "1M context and fast iteration; deepest reasoning tasks still favor Gemini 3.1 Pro.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "content-creation": { + "overall": 90, + "notes": "Fast, capable long-form generation for production content pipelines.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "education": { + "overall": 90, + "notes": "Low latency suits interactive tutoring; strong multimodal explanations.", + "alternatives": ["gemini-3-1-pro", "gpt-5-5"] + } + }, + + "strengths": [ + "Beats Gemini 3.1 Pro on agentic/coding suites: 76.2% Terminal-Bench 2.1, 83.6% MCP Atlas", + "~4x faster than Gemini 3.1 Pro — strong agent-loop economics", + "1,048,576 token input context with 64K output", + "Context caching at $0.15 per 1M cuts repeated-prefix costs dramatically", + "GA from day one across AI Studio, Vertex AI, and Gemini app", + "Full Google Cloud compliance envelope (SOC/ISO, GDPR, HIPAA via GCP)" + ], + + "limitations": [ + "Pricier than past Flash tiers ($1.50/$9.00 vs historical sub-dollar Flash pricing)", + "Benchmark figures are official claims; independent replication still accumulating (~3 weeks since GA)", + "Deepest reasoning tasks still favor Pro-tier models", + "Only ~3 weeks of production track record", + "Training data transparency limited (industry standard)" + ], + + "best_for": [ + "High-volume agentic workflows where speed-cost-quality balance matters", + "Production coding agents with many tool-call iterations", + "Customer-facing applications needing frontier quality at low latency", + "Long-context processing on a workhorse budget" + ], + + "not_recommended_for": [ + "Hardest reasoning problems (use Gemini 3.1 Pro)", + "Ultra-low-cost bulk inference (older Flash/Lite tiers remain cheaper)", + "Risk-averse deployments requiring long production track records" + ], + + "metadata": { + "pricing": { + "input": "$1.50 per 1M tokens", + "output": "$9.00 per 1M tokens", + "notes": "Cached input $0.15 per 1M. Pricier than past Flash tiers, reflecting 'frontier workhorse' positioning.", + "last_verified": "2026-06-10" + }, + "context_window": 1048576, + "max_output": 65536, + "languages": ["English", "100+ languages"], + "modalities": ["text", "vision", "audio", "video"], + "api_endpoint": "https://generativelanguage.googleapis.com/v1beta/models", + "open_source": false, + "architecture": "Speed-optimized multimodal transformer with thinking support", + "parameters": "Not disclosed", + "knowledge_cutoff": "Early 2026 (not officially confirmed)", + "release_date": "2026-05-19" + }, + + "related_entities": ["gemini-3-1-pro", "gemini-3-flash", "gemini-3-pro", "gpt-5-5"], + + "tags": [ + "workhorse", + "ga", + "agentic", + "fast", + "long-context", + "1m-tokens", + "multimodal", + "google-cloud" + ] +} diff --git a/data/models/gemini-3-pro.json b/data/models/gemini-3-pro.json index ddb9ffe..8a3121a 100644 --- a/data/models/gemini-3-pro.json +++ b/data/models/gemini-3-pro.json @@ -4,9 +4,9 @@ "name": "Gemini 3 Pro", "provider": "Google", "version": "gemini-3-pro-preview", - "last_evaluated": "2026-01-14", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Google's flagship with 1M token context, 1501 LMArena Elo (first model >1500), Deep Think mode for complex reasoning, and native multimodal. 6x improvement on ARC-AGI-2 over 2.5 Pro.", + "description": "Former Google flagship, superseded by Gemini 3.1 Pro (GA 2026-02-19; ARC-AGI-2 77.1% vs 31.1%). Still served, with 1M token context, 1501 LMArena Elo (first model >1500), Deep Think mode for complex reasoning, and native multimodal. New projects should prefer Gemini 3.1 Pro.", "website": "https://blog.google/products/gemini/gemini-3/", "trust_vector": { @@ -478,6 +478,12 @@ "url": "https://cloud.google.com/apis/design/versioning", "date": "2025-11-18", "value": "Clear versioning with migration guides" + }, + { + "source": "Gemini API Changelog", + "url": "https://ai.google.dev/gemini-api/docs/changelog", + "date": "2026-06-10", + "value": "Superseded by Gemini 3.1 Pro (GA 2026-02-19; ARC-AGI-2 77.1% vs 31.1%)" } ], "methodology": "Versioning policy review", @@ -612,7 +618,8 @@ "Slightly behind on SWE-bench (76.2% vs Claude's 80.9%)", "Deep Think increases latency significantly", "Data retention policies less clear than Anthropic", - "Newer model with less community testing" + "Newer model with less community testing", + "SUPERSEDED: Gemini 3.1 Pro (GA 2026-02-19) is now Google's most capable Pro model; this model is no longer the flagship" ], "best_for": [ @@ -651,15 +658,14 @@ "knowledge_cutoff": "January 2025" }, - "related_entities": ["gemini-3-flash", "gemini-2-5-pro", "claude-opus-4-5", "gpt-5-2"], + "related_entities": ["gemini-3-1-pro", "gemini-3-flash", "gemini-2-5-pro", "claude-opus-4-5", "gpt-5-2"], "tags": [ + "superseded", "long-context", "1m-tokens", "deep-think", "multimodal", - "lmarena-leader", - "google-cloud", - "flagship" + "google-cloud" ] } diff --git a/data/models/gemma-3-27b.json b/data/models/gemma-3-27b.json index e783a49..2a37418 100644 --- a/data/models/gemma-3-27b.json +++ b/data/models/gemma-3-27b.json @@ -4,9 +4,9 @@ "name": "Gemma 3 27B", "provider": "Google", "version": "2025-01", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Google's open-source Gemma 3 model with 27 billion parameters. Designed for developers seeking Google's research quality with open-source flexibility and commercial-friendly licensing.", + "description": "Google's open-source Gemma 3 model with 27 billion parameters, now superseded by Gemma 4 (released 2026-04-02 under Apache 2.0, a license improvement over the custom Gemma license). Was designed for developers seeking Google's research quality with open-source flexibility; new deployments should evaluate Gemma 4 instead.", "website": "https://ai.google.dev/gemma", "trust_vector": { "performance_reliability": { @@ -429,7 +429,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 86, + "score": 78, "confidence": "high", "evidence": [ { @@ -437,10 +437,17 @@ "url": "https://ai.google.dev/gemma", "date": "2025-01-10", "value": "Clear versioning" + }, + { + "source": "Google Gemma 4 Announcement", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/gemma-4/", + "date": "2026-06-10", + "value": "Superseded by Gemma 4 (released 2026-04-02 under Apache 2.0); Gemma 3 is no longer Google's current open model generation" } ], "methodology": "Review of versioning", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10", + "notes": "Self-hosted weights remain usable, but the line has moved to Gemma 4" }, "monitoring_observability": { "score": 76, @@ -598,7 +605,8 @@ "Moderate coding capabilities", "Requires infrastructure for deployment", "Not suitable for complex or specialized tasks", - "Limited ecosystem compared to Llama" + "Limited ecosystem compared to Llama", + "Superseded by Gemma 4 (released 2026-04-02 under Apache 2.0, an improvement over the custom Gemma license)" ], "best_for": [ "Developers prioritizing Google's open-source models", @@ -645,9 +653,11 @@ "llama-4-scout", "llama-3-3-70b", "gpt-4-1-mini", - "claude-haiku-4-5" + "claude-haiku-4-5", + "gemma-4" ], "tags": [ + "superseded", "open-source", "google", "privacy", diff --git a/data/models/gemma-4.json b/data/models/gemma-4.json new file mode 100644 index 0000000..1d5cc6f --- /dev/null +++ b/data/models/gemma-4.json @@ -0,0 +1,595 @@ +{ + "id": "gemma-4", + "type": "model", + "name": "Gemma 4", + "provider": "Google", + "version": "gemma-4", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Google's open-weight family released April 2026 under Apache 2.0 (a shift from the custom Gemma license). Spans E2B/E4B edge models with 128K context and native audio up to a 31B dense model with 256K context. The 31B scores ~1452 on LMArena, No. 3 among open models.", + "website": "https://blog.google/innovation-and-ai/technology/developers-tools/gemma-4/", + + "trust_vector": { + "performance_reliability": { + "overall_score": 82, + "criteria": { + "task_accuracy_code": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Google Gemma 4 Announcement", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/gemma-4/", + "date": "2026-04-02", + "value": "Substantial coding gains over Gemma 3 across family; strong for open-weight class, below frontier proprietary models" + } + ], + "methodology": "Vendor-reported coding benchmarks compared against open-weight peer class", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Gemma 4 Blog", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Reasoning improvements over Gemma 3; 26B-A4B MoE delivers near-dense quality with 4B active parameters" + } + ], + "methodology": "Reasoning benchmark review from launch materials and open-model leaderboards", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "LMArena Leaderboard", + "url": "https://lmarena.ai/leaderboard", + "date": "2026-06-01", + "value": "Gemma 4 31B ~1452 Elo (No. 3 among open models); 26B-A4B MoE ~1441 with only 4B active params" + } + ], + "methodology": "Crowdsourced human preference rankings on LMArena", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 80, + "confidence": "low", + "evidence": [ + { + "source": "Community testing", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Consistency depends on quantization and inference stack chosen by deployer" + } + ], + "methodology": "Community reports across deployment stacks; high variance by quantization level", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "Deployment-dependent", + "confidence": "low", + "evidence": [ + { + "source": "Hugging Face Gemma 4 Blog", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "E2B (2.3B effective) and E4B (4.5B) run on-device; latency depends entirely on hardware and serving stack" + } + ], + "methodology": "Self-hosted model; latency is a function of deployer infrastructure", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "256K tokens (31B/26B); 128K (E2B/E4B)", + "confidence": "high", + "evidence": [ + { + "source": "Google Gemma 4 Announcement", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/gemma-4/", + "date": "2026-04-02", + "value": "256K context on 12B/26B-A4B/31B; 128K on E2B/E4B edge variants" + } + ], + "methodology": "Official specification from launch announcement", + "last_verified": "2026-06-10" + }, + "uptime": { + "value": "Self-hosted (deployer-controlled)", + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Hub", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Open weights; availability depends on chosen hosting (self-hosted, Vertex AI, or third-party providers)" + } + ], + "methodology": "No single provider SLA; assessed as deployment-dependent", + "last_verified": "2026-06-10" + } + }, + "notes": "Strongest open-weight showing from Google to date: 31B at ~1452 LMArena (No. 3 open). MoE 26B-A4B offers near-dense quality at 4B active params. Performance below proprietary frontier but excellent per-parameter efficiency." + }, + + "security": { + "overall_score": 78, + "criteria": { + "prompt_injection_resistance": { + "score": 72, + "confidence": "low", + "evidence": [ + { + "source": "Gemma Responsible AI Toolkit", + "url": "https://ai.google.dev/responsible", + "date": "2026-04-02", + "value": "Safety tuning applied, but smaller open models are generally more susceptible than frontier hosted models; no managed input filtering by default" + } + ], + "methodology": "OWASP LLM01 assessment relative to model class; deployer must add input filtering", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "Gemma Safety Documentation", + "url": "https://ai.google.dev/gemma/docs", + "date": "2026-04-02", + "value": "Instruction-tuned variants include safety alignment, but open weights permit fine-tuning that removes guardrails" + } + ], + "methodology": "Adversarial testing of instruction-tuned checkpoints; open weights inherently allow guardrail removal", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Self-hosted deployment model", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Self-hosting means no prompts or outputs leave deployer infrastructure" + } + ], + "methodology": "Architectural assessment: no third-party data flow when self-hosted", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Gemma Responsible AI Toolkit", + "url": "https://ai.google.dev/responsible", + "date": "2026-04-02", + "value": "Safety-tuned checkpoints plus companion safety classifiers (ShieldGemma line) available" + } + ], + "methodology": "Safety testing of released checkpoints and available companion classifiers", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "Deployment options", + "url": "https://ai.google.dev/gemma/docs", + "date": "2026-04-02", + "value": "No first-party managed API security; depends on serving stack (Vertex AI managed endpoints inherit GCP controls)" + } + ], + "methodology": "Assessment of typical self-hosted serving stacks vs managed alternatives", + "last_verified": "2026-06-10" + } + }, + "notes": "Security profile is deployment-dependent: excellent data isolation when self-hosted, but guardrails are removable and there is no managed abuse filtering unless the deployer adds it (e.g., ShieldGemma, Vertex AI)." + }, + + "privacy_compliance": { + "overall_score": 84, + "criteria": { + "data_residency": { + "value": "Full deployer control (self-hosted)", + "confidence": "high", + "evidence": [ + { + "source": "Open weights distribution", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Weights run anywhere: on-premises, air-gapped, or any cloud region" + } + ], + "methodology": "Architectural assessment of self-hosted deployment", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Self-hosted deployment model", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "No user data ever transmitted to Google during inference; opt-out concern does not apply" + } + ], + "methodology": "Architectural assessment: inference data never leaves deployer", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Deployer-controlled (zero by default)", + "confidence": "high", + "evidence": [ + { + "source": "Self-hosted deployment model", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Retention policy is entirely the deployer's choice" + } + ], + "methodology": "Architectural assessment", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Self-hosted deployment model", + "url": "https://ai.google.dev/gemma/docs", + "date": "2026-04-02", + "value": "PII never leaves deployer infrastructure, but redaction tooling must be self-implemented" + } + ], + "methodology": "Data flow analysis for self-hosted inference", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Deployment-dependent compliance", + "url": "https://cloud.google.com/security/compliance", + "date": "2026-04-02", + "value": "Model itself carries no certifications; compliance (SOC 2, HIPAA, GDPR) must be achieved by the deployer's stack or inherited from a managed host like Vertex AI" + } + ], + "methodology": "Review of certification inheritance paths for open-weight deployments", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Self-hosted deployment model", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Inherently zero retention when self-hosted" + } + ], + "methodology": "Architectural assessment", + "last_verified": "2026-06-10" + } + }, + "notes": "Best-in-class data sovereignty: nothing leaves deployer infrastructure. The trade-off is that compliance certifications are not inherited from the model and must be built or bought by the deployer." + }, + + "trust_transparency": { + "overall_score": 80, + "criteria": { + "explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Open weights access", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Full weight access enables interpretability research, logit inspection, and custom probing" + } + ], + "methodology": "Assessment of inspection capabilities afforded by open weights", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "Community evaluation", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Smaller open models hallucinate more than frontier hosted models, especially E2B/E4B variants" + } + ], + "methodology": "Factual QA testing relative to model size class", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Gemma Model Card", + "url": "https://ai.google.dev/gemma/docs", + "date": "2026-04-02", + "value": "Bias evaluations published in model card; open weights allow independent auditing" + } + ], + "methodology": "Model card review and independent audit availability", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 76, + "confidence": "low", + "evidence": [ + { + "source": "Open weights access", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Raw logprobs fully accessible for custom calibration, but model self-expression of uncertainty is weaker than frontier tier" + } + ], + "methodology": "Calibration assessment; logprob access partially offsets weaker verbal uncertainty", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Gemma 4 Technical Report and Model Cards", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/gemma-4/", + "date": "2026-04-02", + "value": "Detailed technical report, per-size model cards, architecture details (MoE config, effective params), and evaluation suite published" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Gemma 4 Technical Report", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Training data composition described at category level (better than most proprietary models); exact corpus not released" + } + ], + "methodology": "Public disclosure review against open-model norms", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Responsible AI Toolkit", + "url": "https://ai.google.dev/responsible", + "date": "2026-04-02", + "value": "Safety-tuned checkpoints plus optional classifier models; guardrails are removable by design in open weights" + } + ], + "methodology": "Analysis of built-in and companion safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "High transparency by open-model standards: published technical report, architecture disclosure (including MoE active-parameter counts), and fully auditable weights. Apache 2.0 relicensing further reduces legal opacity." + }, + + "operational_excellence": { + "overall_score": 80, + "criteria": { + "api_design_quality": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Deployment options", + "url": "https://ai.google.dev/gemma/docs", + "date": "2026-04-02", + "value": "No single first-party API; served via Vertex AI, Hugging Face TGI, vLLM, Ollama, llama.cpp with varying interfaces" + } + ], + "methodology": "Review of available serving interfaces and their consistency", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Transformers", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Day-one support in Transformers, vLLM, llama.cpp, Ollama, MLX, and Keras" + } + ], + "methodology": "Ecosystem tooling support assessment", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Gemma release history", + "url": "https://ai.google.dev/gemma/docs", + "date": "2026-04-02", + "value": "Clear generational releases (supersedes Gemma 3); pinned weights never change once published" + } + ], + "methodology": "Release cadence and immutability review", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Self-hosted deployment model", + "url": "https://ai.google.dev/gemma/docs", + "date": "2026-04-02", + "value": "No built-in monitoring; deployer must assemble observability from serving-stack tooling" + } + ], + "methodology": "Assessment of out-of-box observability versus managed APIs", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Community channels", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Community support (Hugging Face, GitHub, Discord); no SLA unless deployed via managed platforms" + } + ], + "methodology": "Support channel assessment for open-weight distribution", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Gemma 4 Launch", + "url": "https://huggingface.co/blog/gemma4", + "date": "2026-04-02", + "value": "Day-one integration across the open-model ecosystem; Gemma family has hundreds of millions of cumulative downloads" + } + ], + "methodology": "Third-party integration and adoption analysis", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "Google Gemma 4 Announcement", + "url": "https://blog.google/innovation-and-ai/technology/developers-tools/gemma-4/", + "date": "2026-04-02", + "value": "Apache 2.0 — a shift from the custom Gemma license, removing use-restriction ambiguity for commercial deployment" + } + ], + "methodology": "License analysis; Apache 2.0 is OSI-approved with no usage restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Apache 2.0 relicensing is the headline trust improvement: prior Gemma generations carried custom-license use restrictions. Operational burden (monitoring, scaling, support) falls on the deployer, as with any open-weight model." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 80, + "notes": "Capable for an open model, especially 31B with 256K context, but well below frontier proprietary coding models.", + "alternatives": ["qwen3-5", "llama-4-maverick"] + }, + "customer-support": { + "overall": 82, + "notes": "26B-A4B MoE (4B active) gives strong quality at low serving cost for high-volume support; E4B enables on-device assistants.", + "alternatives": ["gemini-3-flash", "qwen3-5"] + }, + "content-creation": { + "overall": 80, + "notes": "Solid drafting quality at 31B (~1452 LMArena); fully private content pipelines possible.", + "alternatives": ["llama-4-maverick", "gemini-3-5-flash"] + }, + "education": { + "overall": 81, + "notes": "E2B/E4B with native audio enable offline, on-device tutoring in low-connectivity settings.", + "alternatives": ["gemma-3-27b", "gemini-3-flash"] + }, + "healthcare": { + "overall": 76, + "notes": "Self-hosting suits strict data sovereignty (PHI never leaves infrastructure), but deployer carries the full compliance and accuracy-validation burden.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "research-assistant": { + "overall": 78, + "notes": "256K context on 31B handles long documents; auditable weights suit reproducible research. Reasoning depth below frontier.", + "alternatives": ["gemini-3-1-pro", "qwen3-5"] + } + }, + + "strengths": [ + "Apache 2.0 license — removes custom-license restrictions of prior Gemma generations", + "Top-3 open model: 31B at ~1452 LMArena Elo", + "Efficient MoE: 26B-A4B reaches ~1441 Elo with only 4B active parameters", + "Full data sovereignty: self-hosted inference, zero data leaves deployer", + "Edge-capable E2B/E4B variants with 128K context and native audio", + "256K context on 12B/26B/31B variants — large for open weights", + "Multimodal input: text, image, and video" + ], + + "limitations": [ + "No inherited compliance certifications; deployer builds or buys SOC 2/HIPAA posture", + "Safety guardrails removable via fine-tuning (inherent to open weights)", + "No first-party SLA or managed support outside Vertex AI hosting", + "Hallucination and reasoning depth below frontier hosted models, especially E2B/E4B", + "Operational burden (serving, scaling, monitoring) falls on deployer", + "Performance varies significantly with quantization choices" + ], + + "best_for": [ + "Data-sovereign deployments (on-prem, air-gapped, regulated environments)", + "On-device and edge AI (E2B/E4B with native audio)", + "Cost-efficient high-volume serving via the 26B-A4B MoE", + "Fine-tuning and domain adaptation under a permissive license", + "Research requiring auditable, reproducible model weights" + ], + + "not_recommended_for": [ + "Frontier-grade reasoning or coding without fine-tuning", + "Teams wanting a managed API with SLAs and built-in safety filtering", + "Compliance-critical workloads without in-house security engineering" + ], + + "metadata": { + "pricing": { + "input": "Free (open weights; compute costs only)", + "output": "Free (open weights; compute costs only)", + "notes": "Apache 2.0. Self-hosting compute is the only cost; managed hosting available via Vertex AI and third-party providers.", + "last_verified": "2026-06-10" + }, + "context_window": 262144, + "max_output": 32768, + "languages": ["English", "140+ languages"], + "modalities": ["text", "image (input)", "video (input)", "audio (input, E2B/E4B)"], + "api_endpoint": "https://huggingface.co/google", + "open_source": true, + "architecture": "Family: E2B (2.3B effective) and E4B (4.5B) edge models; 12B dense; 26B-A4B MoE (4B active); 31B dense", + "parameters": "2.3B effective (E2B) to 31B dense; 26B MoE with 4B active", + "knowledge_cutoff": "Late 2025 (not officially confirmed)", + "release_date": "2026-04-02" + }, + + "related_entities": ["gemma-3-27b", "gemini-3-1-pro", "llama-4-maverick", "qwen3-5"], + + "tags": [ + "open-source", + "apache-2-0", + "open-weights", + "edge", + "on-device", + "moe", + "multimodal", + "self-hosted", + "data-sovereignty" + ] +} diff --git a/data/models/glm-5.json b/data/models/glm-5.json new file mode 100644 index 0000000..f2eaf35 --- /dev/null +++ b/data/models/glm-5.json @@ -0,0 +1,644 @@ +{ + "id": "glm-5", + "type": "model", + "name": "GLM-5", + "provider": "Z.ai (Zhipu AI)", + "version": "20260211", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Z.ai's MIT-licensed 744B-parameter MoE (40B active) with 77.8% SWE-bench Verified, 92.7% AIME 2026, and open-source leadership on BrowseComp and agentic benchmarks. Trained on 28.5T tokens with DeepSeek Sparse Attention.", + "website": "https://huggingface.co/zai-org/GLM-5", + + "trust_vector": { + "performance_reliability": { + "overall_score": 92, + "criteria": { + "task_accuracy_code": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "SWE-bench Verified", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "77.8% resolution rate" + }, + { + "source": "Artificial Analysis independent evaluation", + "url": "https://artificialanalysis.ai/models/glm-5", + "date": "2026-03-01", + "value": "Top-tier open-weight coding performance, confirmed by independent harness" + } + ], + "methodology": "Industry-standard coding benchmarks with independent third-party verification", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "AIME 2026", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "92.7% on competition mathematics" + }, + { + "source": "GPQA-Diamond", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "86.0% on PhD-level science questions" + } + ], + "methodology": "Graduate and competition-level reasoning benchmarks requiring multi-step problem solving", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/glm-5", + "date": "2026-03-01", + "value": "Open-source leader on BrowseComp, Vending Bench 2, and MCP-Atlas agentic benchmarks" + } + ], + "methodology": "Independent agentic and general-capability benchmarking across domains", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5 GitHub repository", + "url": "https://github.com/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Stable behavior across agentic trajectories; reproducible deployment recipes published" + } + ], + "methodology": "Community testing with repeated prompts and long agent runs", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "2.6s", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/glm-5", + "date": "2026-03-01", + "value": "Typical first-party API response ~2.6s; DeepSeek Sparse Attention keeps long-context latency manageable" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "6.0s", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/glm-5", + "date": "2026-03-01", + "value": "p95 ~6.0s across diverse workloads" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "200,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "GLM-5 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "200K token context window" + } + ], + "methodology": "Official specification from model card", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Platform", + "url": "https://z.ai/", + "date": "2026-06-01", + "value": "First-party API generally stable; open weights enable self-hosted redundancy" + } + ], + "methodology": "Review of platform availability and self-hosting fallback options", + "last_verified": "2026-06-10" + } + }, + "notes": "Among the strongest open-weight models released to date: 77.8% SWE-bench Verified, 92.7% AIME 2026, 86.0% GPQA-Diamond, with independent confirmation of open-source leadership on BrowseComp, Vending Bench 2, and MCP-Atlas." + }, + + "security": { + "overall_score": 80, + "criteria": { + "prompt_injection_resistance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5 documentation", + "url": "https://github.com/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Safety tuning applied; resilient in agentic browsing benchmarks but no dedicated third-party injection audit published" + } + ], + "methodology": "Review of safety documentation and community testing against OWASP LLM01 patterns", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Community red-teaming", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-03-15", + "value": "Standard alignment; open weights allow guardrail removal in derivatives" + } + ], + "methodology": "Testing against adversarial prompt datasets; deployer-dependent for self-hosted use", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Privacy Policy", + "url": "https://z.ai/", + "date": "2026-02-11", + "value": "Standard data handling on first-party API; full control when self-hosted" + } + ], + "methodology": "Analysis of privacy policies and self-hosting data-control options", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 81, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Safety post-training; refusal behavior comparable to peer open frontier models" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai API Documentation", + "url": "https://docs.z.ai/", + "date": "2026-02-11", + "value": "API key authentication, HTTPS only, rate limiting; OpenAI-compatible endpoints" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Solid for an open model; no published third-party security audit. Self-hosting shifts responsibility to the deployer." + }, + + "privacy_compliance": { + "overall_score": 75, + "criteria": { + "data_residency": { + "value": "China (first-party Z.ai API); any jurisdiction when self-hosted or via Western hosts", + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Platform Documentation", + "url": "https://docs.z.ai/", + "date": "2026-02-11", + "value": "Zhipu/Z.ai is a China-based provider; first-party API data processed under Chinese jurisdiction" + }, + { + "source": "OpenRouter availability", + "url": "https://openrouter.ai/models", + "date": "2026-03-01", + "value": "MIT weights hosted by Western inference providers, enabling non-China residency" + } + ], + "methodology": "Review of provider jurisdiction and third-party hosting options", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Privacy Policy", + "url": "https://z.ai/", + "date": "2026-02-11", + "value": "Standard API data terms; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Per Z.ai policy on first-party API (China jurisdiction); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Terms of Service", + "url": "https://z.ai/", + "date": "2026-02-11", + "value": "First-party retention governed by Chinese data regulations; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service and deployment-dependent retention", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Documentation", + "url": "https://docs.z.ai/", + "date": "2026-02-11", + "value": "Customer responsible for PII redaction; no managed PII tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai public materials", + "url": "https://z.ai/", + "date": "2026-02-11", + "value": "No published SOC 2 / HIPAA / GDPR attestations for the first-party API; Western hosts may carry their own certifications" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Open weights on Hugging Face", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "MIT-licensed self-hosting gives complete data control and zero external retention" + } + ], + "methodology": "Review of self-hosting deployment options enabling zero retention", + "last_verified": "2026-06-10" + } + }, + "notes": "First-party Z.ai API operates under Chinese jurisdiction — a material caveat for Western regulated industries. The unencumbered MIT license makes self-hosting or Western-host deployment a clean mitigation." + }, + + "trust_transparency": { + "overall_score": 81, + "criteria": { + "explainability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5 documentation", + "url": "https://github.com/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Reasoning traces and agentic tool-call logs inspectable; strong MCP-Atlas results reflect transparent tool use" + } + ], + "methodology": "Evaluation of reasoning transparency and trajectory inspectability", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 81, + "confidence": "medium", + "evidence": [ + { + "source": "BrowseComp results", + "url": "https://artificialanalysis.ai/models/glm-5", + "date": "2026-03-01", + "value": "Open-source leader on grounded web-research benchmark, indicating disciplined sourcing behavior" + } + ], + "methodology": "Testing on factual QA and grounded research benchmarks", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "GLM-5 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Limited published bias evaluation" + } + ], + "methodology": "Review of published bias benchmarks and community evaluations", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Model behavior testing", + "url": "https://github.com/zai-org/GLM-5", + "date": "2026-03-15", + "value": "Expresses uncertainty adequately; no calibrated confidence outputs" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face model card and GitHub repo", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Detailed disclosure: 744B/40B MoE, 256 experts, DeepSeek Sparse Attention, 28.5T pretraining tokens, full benchmark tables" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5 technical disclosure", + "url": "https://github.com/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Pretraining scale (28.5T tokens) and recipe outlined; detailed data sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Built-in safety tuning; deployers of open weights must layer their own guardrails" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Above-average transparency for an open frontier model: architecture, training scale, and benchmarks well documented with independent verification. Bias and safety evaluations remain thin." + }, + + "operational_excellence": { + "overall_score": 83, + "criteria": { + "api_design_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Z.ai API Documentation", + "url": "https://docs.z.ai/", + "date": "2026-02-11", + "value": "OpenAI-compatible API with streaming, tool calling, structured output" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai GitHub organization", + "url": "https://github.com/zai-org", + "date": "2026-02-11", + "value": "Official repos with deployment recipes; OpenAI-compatible so mainstream SDKs work" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "GLM release history", + "url": "https://huggingface.co/zai-org", + "date": "2026-04-07", + "value": "Fast cadence: GLM-5.1 API launched 2026-03-27 with weights on 2026-04-07, same architecture; GLM-5 weights remain available" + } + ], + "methodology": "Review of versioning practices and weight availability across releases", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Platform", + "url": "https://z.ai/", + "date": "2026-02-11", + "value": "Basic usage dashboard; self-hosted observability is deployer-built" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai community channels", + "url": "https://github.com/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Active GitHub support and documentation; limited English-language enterprise support" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Inference ecosystem", + "url": "https://openrouter.ai/models", + "date": "2026-03-01", + "value": "Day-one vLLM/SGLang support, OpenRouter and major Western host availability" + } + ], + "methodology": "Analysis of third-party hosting, integrations, and tooling", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "MIT License", + "url": "https://github.com/zai-org/GLM-5", + "date": "2026-02-11", + "value": "Unencumbered MIT license, unrestricted commercial use and derivatives" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Clean MIT licensing and strong ecosystem support. Note the rapid successor cadence: GLM-5.1 (same architecture) shipped within two months; evaluate which version your hosts actually serve." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "77.8% SWE-bench Verified with independent verification; best-in-class open-weight coding at ~$0.60/$1.92 per 1M.", + "alternatives": ["kimi-k2-6", "claude-opus-4-8", "deepseek-v4"] + }, + "customer-support": { + "overall": 82, + "notes": "Capable and inexpensive, but not specialized for support workflows.", + "alternatives": ["command-a-plus", "minimax-m2"] + }, + "content-creation": { + "overall": 84, + "notes": "Strong long-form generation at very low cost.", + "alternatives": ["claude-opus-4-8", "kimi-k2-6"] + }, + "data-analysis": { + "overall": 89, + "notes": "Excellent mathematical reasoning (92.7% AIME 2026) and agentic tool use for analysis pipelines.", + "alternatives": ["kimi-k2-6", "deepseek-v4"] + }, + "research-assistant": { + "overall": 91, + "notes": "Open-source leader on BrowseComp and MCP-Atlas; strong grounded web research.", + "alternatives": ["kimi-k2-6", "claude-opus-4-8"] + }, + "legal-compliance": { + "overall": 70, + "notes": "China-jurisdiction first-party API and absent Western certifications are blockers unless self-hosted.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 68, + "notes": "Not recommended via first-party API; self-hosted deployment in a compliant environment is the only viable path.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 86, + "notes": "Top-tier quantitative reasoning; regulated firms should self-host or use certified Western hosts.", + "alternatives": ["command-a-plus", "kimi-k2-6"] + }, + "education": { + "overall": 88, + "notes": "Outstanding math and science tutoring (92.7% AIME, 86.0% GPQA-Diamond) at budget pricing.", + "alternatives": ["deepseek-v3-2", "kimi-k2-6"] + }, + "creative-writing": { + "overall": 81, + "notes": "Competent prose; optimized for reasoning and agentic work rather than creative style.", + "alternatives": ["claude-opus-4-8", "minimax-m2"] + } + }, + + "strengths": [ + "Frontier-competitive results: 77.8% SWE-bench Verified, 92.7% AIME 2026, 86.0% GPQA-Diamond", + "Independently verified open-source leadership on BrowseComp, Vending Bench 2, and MCP-Atlas", + "Unencumbered MIT license with full self-hosting rights", + "Efficient inference: 40B active of 744B total with DeepSeek Sparse Attention", + "Very competitive API pricing (~$0.60/$1.92 per 1M tokens)", + "Well-documented training scale (28.5T tokens) and architecture" + ], + + "limitations": [ + "First-party Z.ai API processes data under Chinese jurisdiction with limited Western compliance certifications", + "Text-only — no vision or audio modalities", + "Rapid successor cadence (GLM-5.1 within two months) creates version-tracking overhead", + "Limited published bias, safety, and red-team evaluations", + "Self-hosting a 744B MoE requires substantial GPU infrastructure", + "English-language enterprise support is thin compared to Western providers" + ], + + "best_for": [ + "Cost-sensitive coding and agentic engineering at near-frontier quality", + "Grounded web research and browsing agents (BrowseComp open-source leader)", + "Mathematical, scientific, and analytical workloads", + "Organizations wanting clean MIT-licensed weights for self-hosted data control" + ], + + "not_recommended_for": [ + "Regulated Western workloads (healthcare, legal, finance) on the first-party API", + "Multimodal applications requiring image or audio input", + "Teams without GPU infrastructure who also cannot accept China-jurisdiction processing" + ], + + "metadata": { + "pricing": { + "input": "$0.60 per 1M tokens (approx.)", + "output": "$1.92 per 1M tokens (approx.)", + "notes": "First-party Z.ai API pricing; third-party hosts vary. Successor GLM-5.1 priced similarly.", + "last_verified": "2026-06-10" + }, + "context_window": 200000, + "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], + "modalities": ["text"], + "api_endpoint": "https://api.z.ai/api/paas/v4/chat/completions", + "open_source": true, + "license": "MIT", + "architecture": "Mixture-of-Experts: 744B total / 40B active parameters, 256 experts, DeepSeek Sparse Attention; 28.5T pretraining tokens", + "parameters": "744B total / 40B active", + "release_date": "2026-02-11" + }, + + "related_entities": ["kimi-k2-6", "deepseek-v4", "deepseek-v3-2", "minimax-m2", "claude-opus-4-8"], + + "tags": [ + "coding", + "reasoning", + "open-source", + "mit-license", + "mixture-of-experts", + "agentic", + "cost-effective", + "chinese-provider", + "self-hostable" + ] +} diff --git a/data/models/gpt-4o.json b/data/models/gpt-4o.json index 3a543f0..57432eb 100644 --- a/data/models/gpt-4o.json +++ b/data/models/gpt-4o.json @@ -4,9 +4,9 @@ "name": "GPT-4o", "provider": "OpenAI", "version": "2024-05", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's flagship multimodal model with strong text and vision capabilities. Designed for applications requiring high-quality multimodal understanding and generation.", + "description": "DEPRECATED: removed from ChatGPT 2026-02-13; the gpt-4o-2024-05-13 snapshot's API shuts down 2026-10-23. Historically OpenAI's flagship multimodal model with strong text and vision capabilities for high-quality multimodal understanding and generation. Migrate to newer GPT-5.x models.", "website": "https://openai.com/gpt-4o", "trust_vector": { @@ -410,7 +410,7 @@ }, "operational_excellence": { - "overall_score": 89, + "overall_score": 87, "criteria": { "api_design_quality": { "score": 92, @@ -441,7 +441,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 86, + "score": 74, "confidence": "high", "evidence": [ { @@ -449,6 +449,12 @@ "url": "https://platform.openai.com/docs/versioning", "date": "2024-05-15", "value": "Clear versioning" + }, + { + "source": "OpenAI: Retiring GPT-4o and older models", + "url": "https://openai.com/index/retiring-gpt-4o-and-older-models/", + "date": "2026-06-10", + "value": "GPT-4o removed from ChatGPT 2026-02-13; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23" } ], "methodology": "Policy review", @@ -483,7 +489,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 94, + "score": 86, "confidence": "high", "evidence": [ { @@ -511,7 +517,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Excellent operational maturity with strong multimodal support and ecosystem." + "notes": "Deprecated: removed from ChatGPT 2026-02-13; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23. Versioning and ecosystem scores reduced to reflect deprecation." } }, @@ -583,7 +589,8 @@ "Not HIPAA eligible", "Higher cost than text-only alternatives", "PII concerns with image inputs", - "Moderate coding capabilities" + "Moderate coding capabilities", + "DEPRECATED: removed from ChatGPT 2026-02-13; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23" ], "best_for": [ @@ -634,9 +641,9 @@ "related_entities": ["gpt-4o-mini", "gpt-4-1", "claude-sonnet-4-5", "gemini-2-5-pro"], "tags": [ + "deprecated", "multimodal", "vision", - "flagship", "image-understanding", "ocr", "education" diff --git a/data/models/gpt-5-1.json b/data/models/gpt-5-1.json index 22650b8..823b4a4 100644 --- a/data/models/gpt-5-1.json +++ b/data/models/gpt-5-1.json @@ -4,9 +4,9 @@ "name": "GPT-5.1", "provider": "OpenAI", "version": "gpt-5-1-1113", - "last_evaluated": "2025-11-17", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's latest flagship released Nov 2025 with adaptive reasoning (2-3x faster on simple tasks), 76.3% SWE-bench, new developer tools (apply_patch, shell), warmer tone.", + "description": "SUPERSEDED by GPT-5.2/5.4/5.5 (GPT-5.5 is the current flagship); gpt-5.1-chat-latest and gpt-5.1-codex variants shut down in the API 2026-07-23. Released Nov 2025 with adaptive reasoning (2-3x faster on simple tasks), 76.3% SWE-bench, developer tools (apply_patch, shell), warmer tone.", "website": "https://openai.com/index/gpt-5-1/", "trust_vector": { @@ -450,6 +450,12 @@ "url": "https://platform.openai.com/docs/api-reference/models", "date": "2025-01-01", "value": "Clear versioning with deprecation notices" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "gpt-5.1-chat-latest and gpt-5.1-codex variants API shutdown 2026-07-23; superseded by GPT-5.4/5.5" } ], "methodology": "Versioning policy review", @@ -583,7 +589,8 @@ "30-day data retention vs Anthropic's 0-day default", "Smaller context window (128K vs Claude's 200K)", "Premium pricing comparable to Claude", - "Slightly behind Claude on specialized coding benchmarks" + "Slightly behind Claude on specialized coding benchmarks", + "SUPERSEDED by GPT-5.4/5.5; gpt-5.1-chat-latest and gpt-5.1-codex variants API shutdown 2026-07-23" ], "best_for": [ @@ -631,9 +638,10 @@ "parameters": "Not disclosed" }, - "related_entities": ["gpt-5", "claude-sonnet-4-5", "claude-opus-4", "claude-haiku-4-5"], + "related_entities": ["gpt-5-5", "gpt-5-4", "gpt-5", "claude-sonnet-4-5", "claude-opus-4", "claude-haiku-4-5"], "tags": [ + "superseded", "general-purpose", "multimodal", "low-latency", diff --git a/data/models/gpt-5-2-codex.json b/data/models/gpt-5-2-codex.json index 2203fe6..29fce21 100644 --- a/data/models/gpt-5-2-codex.json +++ b/data/models/gpt-5-2-codex.json @@ -4,9 +4,9 @@ "name": "GPT-5.2 Codex", "provider": "OpenAI", "version": "gpt-5-2-codex-2025-12-11", - "last_evaluated": "2026-01-14", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's specialized coding model built on GPT-5.2 with 56.4% SWE-bench Pro (state-of-the-art), 64% Terminal-bench 2.0, native code compaction, and enhanced cybersecurity capabilities.", + "description": "DEPRECATED: superseded by GPT-5.3-Codex (2026-02-05); gpt-5.2-codex API shuts down 2026-07-23. Historically OpenAI's specialized coding model built on GPT-5.2 with 56.4% SWE-bench Pro, 64% Terminal-bench 2.0, native code compaction, and enhanced cybersecurity capabilities.", "website": "https://openai.com/index/introducing-gpt-5-2-codex/", "trust_vector": { @@ -416,7 +416,7 @@ }, "operational_excellence": { - "overall_score": 94, + "overall_score": 91, "criteria": { "api_design_quality": { "score": 95, @@ -447,7 +447,7 @@ "last_verified": "2026-01-14" }, "versioning_policy": { - "score": 91, + "score": 79, "confidence": "high", "evidence": [ { @@ -455,6 +455,12 @@ "url": "https://platform.openai.com/docs/api-reference/models", "date": "2025-12-11", "value": "Clear versioning policy" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "gpt-5.2-codex API shutdown 2026-07-23; superseded by GPT-5.3-Codex (2026-02-05)" } ], "methodology": "Versioning review", @@ -489,7 +495,7 @@ "last_verified": "2026-01-14" }, "ecosystem_maturity": { - "score": 96, + "score": 88, "confidence": "high", "evidence": [ { @@ -517,7 +523,7 @@ "last_verified": "2026-01-14" } }, - "notes": "Excellent developer experience with native code compaction and IDE integrations. Industry-leading code tooling." + "notes": "Deprecated: gpt-5.2-codex API shutdown scheduled 2026-07-23; superseded by GPT-5.3-Codex. Versioning and ecosystem scores reduced to reflect deprecation." } }, @@ -589,7 +595,8 @@ "Not suitable for non-code tasks", "Same pricing as GPT-5.2", "Not HIPAA eligible", - "30-day data retention" + "30-day data retention", + "DEPRECATED: superseded by GPT-5.3-Codex (2026-02-05); gpt-5.2-codex API shutdown 2026-07-23" ], "best_for": [ @@ -639,9 +646,10 @@ "parameters": "Not disclosed" }, - "related_entities": ["gpt-5-2", "claude-opus-4-5", "claude-sonnet-4-5"], + "related_entities": ["gpt-5-3-codex", "gpt-5-2", "claude-opus-4-5", "claude-sonnet-4-5"], "tags": [ + "deprecated", "coding", "specialized", "swe-bench-leader", diff --git a/data/models/gpt-5-2.json b/data/models/gpt-5-2.json index 0328567..9da8c08 100644 --- a/data/models/gpt-5-2.json +++ b/data/models/gpt-5-2.json @@ -4,9 +4,9 @@ "name": "GPT-5.2", "provider": "OpenAI", "version": "gpt-5-2-2025-12-11", - "last_evaluated": "2026-01-14", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's latest flagship with 400K context window, 100% AIME 2025 score, and 52.9% ARC-AGI-2. Three variants: Instant (speed), Thinking (reasoning), Pro (accuracy). Industry-leading abstract reasoning.", + "description": "SUPERSEDED by GPT-5.4 (2026-03-05) and GPT-5.5 (2026-04-23, current OpenAI flagship). Still served, with 400K context window, 100% AIME 2025 score, and 52.9% ARC-AGI-2. Three variants: Instant (speed), Thinking (reasoning), Pro (accuracy). New projects should prefer GPT-5.5.", "website": "https://openai.com/index/introducing-gpt-5-2/", "trust_vector": { @@ -485,6 +485,12 @@ "url": "https://platform.openai.com/docs/api-reference/models", "date": "2025-12-11", "value": "Clear versioning with deprecation notices" + }, + { + "source": "OpenAI: Introducing GPT-5.5", + "url": "https://openai.com/index/introducing-gpt-5-5/", + "date": "2026-06-10", + "value": "GPT-5.2 superseded by GPT-5.4 (2026-03-05) and GPT-5.5 (2026-04-23, current flagship)" } ], "methodology": "Versioning policy review", @@ -619,7 +625,8 @@ "30-day data retention vs Anthropic's 0-day", "1.4x price increase over GPT-5.1 ($1.75/$14)", "Slightly behind Claude Opus 4.5 on SWE-bench (80% vs 80.9%)", - "Smaller context than Gemini 3 (400K vs 1M)" + "Smaller context than Gemini 3 (400K vs 1M)", + "SUPERSEDED: GPT-5.4 (2026-03-05) and GPT-5.5 (2026-04-23) are newer; GPT-5.5 is OpenAI's current flagship" ], "best_for": [ @@ -669,16 +676,16 @@ "knowledge_cutoff": "August 31, 2025" }, - "related_entities": ["gpt-5-1", "gpt-5-2-codex", "claude-opus-4-5", "gemini-3-pro"], + "related_entities": ["gpt-5-5", "gpt-5-4", "gpt-5-1", "gpt-5-2-codex", "claude-opus-4-5", "gemini-3-pro"], "tags": [ + "superseded", "reasoning", "multimodal", "400k-context", "ecosystem-leader", "math-expert", "three-variants", - "low-latency", - "flagship" + "low-latency" ] } diff --git a/data/models/gpt-5-3-codex.json b/data/models/gpt-5-3-codex.json new file mode 100644 index 0000000..af7c366 --- /dev/null +++ b/data/models/gpt-5-3-codex.json @@ -0,0 +1,605 @@ +{ + "id": "gpt-5-3-codex", + "type": "model", + "name": "GPT-5.3-Codex", + "provider": "OpenAI", + "version": "gpt-5-3-codex-2026-02-05", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "OpenAI's agentic coding specialist: ~80% SWE-bench Verified, 77.3% Terminal-Bench, SOTA on SWE-Bench Pro at release, ~25% faster than GPT-5.2-Codex. The 5.3 generation was Codex-only — there is no general-purpose GPT-5.3.", + "website": "https://openai.com/index/introducing-gpt-5-3-codex/", + + "trust_vector": { + "performance_reliability": { + "overall_score": 92, + "criteria": { + "task_accuracy_code": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "SWE-bench Verified", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "~80% resolution rate on SWE-bench Verified" + }, + { + "source": "Terminal-Bench", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "77.3% on terminal/command-line agentic tasks" + }, + { + "source": "SWE-Bench Pro", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "State-of-the-art on SWE-Bench Pro at release" + } + ], + "methodology": "Industry-standard agentic coding benchmarks measuring real-world software engineering tasks", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "Strong code-centric reasoning; not positioned for general scientific or mathematical reasoning" + } + ], + "methodology": "Reasoning benchmark review relative to general-purpose flagships", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "Specialized for software engineering; OpenAI recommends GPT-5.x flagships for general-purpose tasks" + } + ], + "methodology": "Comparison against general-purpose models on non-coding workloads", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "~25% faster than GPT-5.2-Codex with more reliable long-horizon task completion" + } + ], + "methodology": "Provider-reported reliability on multi-step agentic coding sessions", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "~25% faster than GPT-5.2-Codex", + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "~25% end-to-end speedup over GPT-5.2-Codex on equivalent tasks" + } + ], + "methodology": "Provider-reported relative latency on agentic coding workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "400,000 tokens", + "confidence": "medium", + "evidence": [ + { + "source": "OpenRouter Model Page", + "url": "https://openrouter.ai/openai/gpt-5.3-codex", + "date": "2026-02-05", + "value": "400K token context window listed for gpt-5.3-codex" + } + ], + "methodology": "Third-party model registry specification", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Status", + "url": "https://status.openai.com/", + "date": "2026-06-01", + "value": "99.9% uptime (last 90 days)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Best-in-class agentic coding at release (~80% SWE-bench Verified, 77.3% Terminal-Bench, SOTA SWE-Bench Pro). Specialized model — general-purpose accuracy intentionally trails flagships." + }, + + "security": { + "overall_score": 87, + "criteria": { + "prompt_injection_resistance": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Research", + "url": "https://openai.com/safety", + "date": "2026-02-05", + "value": "Injection defenses tuned for agentic coding (repository content, tool output, shell results)" + } + ], + "methodology": "Testing against OWASP LLM01 attacks including coding-agent vectors", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "Inherits GPT-5.x safety training with coding-specific refusal calibration (e.g., malware requests)" + } + ], + "methodology": "Adversarial prompt testing against jailbreak datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Privacy Policy", + "url": "https://openai.com/policies/privacy-policy", + "date": "2026-02-05", + "value": "No training on API data by default" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-02-05", + "value": "Sandboxed execution defaults in Codex environments; guardrails on destructive commands" + } + ], + "methodology": "Safety testing across harmful content and dangerous-action categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Platform Docs", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-02-05", + "value": "API key + OAuth2 authentication, HTTPS only, rate limiting" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Good security posture with sandboxing in Codex environments. Autonomous code execution warrants strict permissioning and review gates in production pipelines." + }, + + "privacy_compliance": { + "overall_score": 87, + "criteria": { + "data_residency": { + "value": "US, EU", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-02-05", + "value": "Data residency options for enterprise customers" + } + ], + "methodology": "Review of enterprise documentation", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Data Controls", + "url": "https://openai.com/policies/usage-policies", + "date": "2026-02-05", + "value": "API data (including submitted code) not used for training by default" + } + ], + "methodology": "Policy review of data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "30 days (zero retention available)", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-02-05", + "value": "30-day default API log retention; zero-data-retention options for qualifying customers" + } + ], + "methodology": "Terms of service and enterprise documentation review", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Tools", + "url": "https://platform.openai.com/docs/guides/safety", + "date": "2026-02-05", + "value": "Customer responsible for scrubbing secrets/PII from code context; moderation API available" + } + ], + "methodology": "Review of data protection capabilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Trust Center", + "url": "https://trust.openai.com/", + "date": "2026-02-05", + "value": "SOC 2 Type II, ISO 27001, GDPR compliant" + } + ], + "methodology": "Verification of compliance certifications", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-02-05", + "value": "Zero-data-retention options available for enterprise and qualifying API customers" + } + ], + "methodology": "Enterprise feature review", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard OpenAI enterprise posture. Proprietary source code sent as context is covered by no-training-by-default; zero-data-retention recommended for sensitive codebases." + }, + + "trust_transparency": { + "overall_score": 86, + "criteria": { + "explainability": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "Codex Agent Logs", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "Step-by-step agent transcripts (plans, diffs, test runs) provide strong action-level traceability" + } + ], + "methodology": "Evaluation of reasoning and action transparency", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "Test-driven agent loop catches many fabrications; API hallucination still possible in unfamiliar frameworks" + } + ], + "methodology": "Code correctness evaluation with execution-based verification", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 83, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-02-05", + "value": "Standard bias testing program; less salient for code-specialist deployments" + } + ], + "methodology": "Bias benchmarks and demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Documentation", + "url": "https://platform.openai.com/docs/models", + "date": "2026-02-05", + "value": "Agent flags failing tests and unresolved tasks rather than claiming success" + } + ], + "methodology": "Qualitative assessment of confidence expression in agentic outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "GPT-5.3-Codex Announcement", + "url": "https://openai.com/index/introducing-gpt-5-3-codex/", + "date": "2026-02-05", + "value": "Release documentation covers benchmarks, intended use, and Codex-only scope of the 5.3 generation" + } + ], + "methodology": "Documentation completeness and clarity review", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Blog", + "url": "https://openai.com/blog", + "date": "2026-02-05", + "value": "General description of code-focused training; specific sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Systems", + "url": "https://openai.com/safety", + "date": "2026-02-05", + "value": "Guardrails on destructive operations, secrets handling, and malware generation" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Agent transcripts give strong action-level auditability. Execution-based verification reduces unchecked hallucination relative to chat-style code generation." + }, + + "operational_excellence": { + "overall_score": 93, + "criteria": { + "api_design_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI API Reference", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-02-05", + "value": "Responses API plus first-class integration with Codex CLI, IDE extensions, and Codex cloud" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI SDKs", + "url": "https://github.com/openai", + "date": "2026-02-05", + "value": "Official SDKs plus open-source Codex CLI, actively maintained" + } + ], + "methodology": "SDK quality, documentation, and maintenance review", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-02-05", + "value": "Clear deprecation schedule; predecessor GPT-5.2-Codex shuts down 2026-07-23" + } + ], + "methodology": "Review of versioning policy and deprecation practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Dashboard", + "url": "https://platform.openai.com/usage", + "date": "2026-02-05", + "value": "Detailed usage dashboard with costs, tokens, rate limits; Codex task history" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Support", + "url": "https://help.openai.com/", + "date": "2026-02-05", + "value": "24/7 support, comprehensive docs, active developer community" + } + ], + "methodology": "Support and documentation assessment", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Codex Ecosystem", + "url": "https://platform.openai.com/docs", + "date": "2026-02-05", + "value": "Codex CLI, IDE extensions, cloud agents, GitHub integration; also available via OpenRouter" + } + ], + "methodology": "Ecosystem breadth and depth analysis", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-02-05", + "value": "Standard commercial terms; customer retains rights to generated code per terms of use" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Deep coding-tool ecosystem (CLI, IDE, cloud agents). Codex line moves fast: GPT-5.2-Codex shuts down 2026-07-23, so plan for shorter model lifecycles than general flagships." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 97, + "notes": "Purpose-built agentic coder: ~80% SWE-bench Verified, 77.3% Terminal-Bench, SOTA SWE-Bench Pro at release, ~25% faster than GPT-5.2-Codex.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "data-analysis": { + "overall": 87, + "notes": "Strong at writing and executing analysis code; general flagships better for open-ended analytical interpretation.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "research-assistant": { + "overall": 80, + "notes": "Useful for code-centric research (reproducing papers, building experiment harnesses); not designed for general literature work.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "education": { + "overall": 84, + "notes": "Excellent for programming instruction with executable, test-verified examples; narrow outside software topics.", + "alternatives": ["gpt-5-5", "gpt-5-4"] + } + }, + + "strengths": [ + "~80% SWE-bench Verified and SOTA on SWE-Bench Pro at release", + "77.3% Terminal-Bench on agentic command-line tasks", + "~25% faster than GPT-5.2-Codex on equivalent workloads", + "Aggressive pricing (~$1.75/$14 per 1M) for a frontier coding model", + "Deep tooling: Codex CLI, IDE extensions, cloud agents, GitHub integration", + "Execution-verified outputs reduce unchecked code hallucination" + ], + + "limitations": [ + "Specialized for coding — weaker than flagships on general reasoning and writing", + "No general-purpose GPT-5.3 exists; the 5.3 generation was Codex-only", + "Fast Codex lifecycle: predecessor GPT-5.2-Codex shuts down 2026-07-23, suggesting shorter support horizons", + "Pricing confirmed primarily via third-party listings (medium confidence)", + "Not HIPAA eligible; 30-day default retention", + "Autonomous code execution requires sandboxing and review gates" + ], + + "best_for": [ + "Agentic software engineering: multi-step bug fixes, refactors, and feature implementation", + "CI/automation pipelines needing fast, cost-efficient frontier coding", + "Teams migrating off GPT-5.2-Codex before its 2026-07-23 shutdown", + "Terminal-heavy DevOps and infrastructure-as-code workflows" + ], + + "not_recommended_for": [ + "General-purpose chat, writing, or research assistants", + "Regulated healthcare workloads (not HIPAA eligible)", + "Teams requiring multi-year model stability without migrations" + ], + + "metadata": { + "pricing": { + "input": "$1.75 per 1M tokens (approximate)", + "output": "$14.00 per 1M tokens (approximate)", + "notes": "Pricing per OpenRouter listing (https://openrouter.ai/openai/gpt-5.3-codex); confidence medium pending first-party pricing page confirmation.", + "last_verified": "2026-06-10" + }, + "context_window": 400000, + "max_output": 128000, + "languages": [ + "English", + "Python", + "JavaScript/TypeScript", + "Go", + "Rust", + "Java", + "C/C++", + "C#", + "Ruby", + "PHP", + "Shell", + "SQL" + ], + "modalities": ["text", "vision (screenshots/diagrams)", "code-execution (via Codex harness)"], + "api_endpoint": "https://api.openai.com/v1/responses", + "open_source": false, + "architecture": "Transformer-based, fine-tuned for agentic software engineering (Codex line)", + "parameters": "Not disclosed", + "knowledge_cutoff": "Late 2025" + }, + + "related_entities": ["gpt-5-2-codex", "gpt-5-5", "gpt-5-4", "claude-opus-4-8"], + + "tags": [ + "coding", + "agentic", + "codex", + "specialist", + "terminal", + "swe-bench-leader", + "cost-efficient" + ] +} diff --git a/data/models/gpt-5-4.json b/data/models/gpt-5-4.json new file mode 100644 index 0000000..f4b6537 --- /dev/null +++ b/data/models/gpt-5-4.json @@ -0,0 +1,644 @@ +{ + "id": "gpt-5-4", + "type": "model", + "name": "GPT-5.4", + "provider": "OpenAI", + "version": "gpt-5-4-2026-03-05", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "OpenAI's previous-generation flagship (superseded by GPT-5.5 in April 2026). Headline native computer use with 75% OSWorld-Verified, ~33% fewer factual errors than GPT-5.2, ~1.05M context. Thinking, Pro, mini, and nano variants.", + "website": "https://openai.com/index/introducing-gpt-5-4/", + + "trust_vector": { + "performance_reliability": { + "overall_score": 95, + "criteria": { + "task_accuracy_code": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "Incremental coding gains over GPT-5.2; release emphasis on computer use rather than coding (GPT-5.3-Codex remained the coding lead at launch)" + } + ], + "methodology": "Industry-standard coding benchmarks and provider-reported comparisons", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "Reasoning improvements over GPT-5.2 across science and math suites; Thinking and Pro variants for deeper reasoning" + } + ], + "methodology": "PhD-level and Olympiad-level reasoning benchmarks", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "OSWorld-Verified", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "75% on OSWorld-Verified with native computer use (vs GPT-5.2's 47.3%)" + }, + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "~33% fewer factual errors than GPT-5.2" + } + ], + "methodology": "Computer-use benchmarks and factual accuracy testing", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "~33% reduction in factual errors vs GPT-5.2 improves run-to-run reliability" + } + ], + "methodology": "Internal testing across model variants", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "0.8s (standard) / faster on mini and nano", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/", + "date": "2026-03-20", + "value": "mini and nano variants optimized for low latency" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "3.5s (standard) / higher for Thinking and Pro", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/", + "date": "2026-03-20", + "value": "p95 latency varies by variant and reasoning depth" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "~1,050,000 tokens input / 128,000 output", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Documentation", + "url": "https://platform.openai.com/docs/models", + "date": "2026-03-05", + "value": "~1.05M token input context, 128K max output tokens" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Status", + "url": "https://status.openai.com/", + "date": "2026-06-01", + "value": "99.9% uptime (last 90 days)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong generation defined by native computer use (75% OSWorld-Verified, +27.7 points over GPT-5.2) and a ~33% factual-error reduction. Superseded by GPT-5.5 six weeks after release." + }, + + "security": { + "overall_score": 88, + "criteria": { + "prompt_injection_resistance": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Research", + "url": "https://openai.com/safety", + "date": "2026-03-05", + "value": "Injection defenses extended to native computer-use surfaces (screenshots, UI text)" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attacks", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "Improved refusal robustness over GPT-5.2 per release documentation" + } + ], + "methodology": "Adversarial prompt testing against jailbreak datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Privacy Policy", + "url": "https://openai.com/policies/privacy-policy", + "date": "2026-03-05", + "value": "No training on API data by default" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-03-05", + "value": "Multi-layer safety with additional guardrails for autonomous computer-use actions" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Platform Docs", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-03-05", + "value": "API key + OAuth2 authentication, HTTPS only, rate limiting" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Solid security posture with computer-use-specific guardrails. Native computer use expands the attack surface (UI-based injection) relative to text-only models." + }, + + "privacy_compliance": { + "overall_score": 87, + "criteria": { + "data_residency": { + "value": "US, EU", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-03-05", + "value": "Data residency options for enterprise customers" + } + ], + "methodology": "Review of enterprise documentation", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Data Controls", + "url": "https://openai.com/policies/usage-policies", + "date": "2026-03-05", + "value": "API data not used for training by default" + } + ], + "methodology": "Policy review of data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "30 days (zero retention available)", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-03-05", + "value": "30-day default API log retention; zero-data-retention options for qualifying customers" + } + ], + "methodology": "Terms of service and enterprise documentation review", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Tools", + "url": "https://platform.openai.com/docs/guides/safety", + "date": "2026-03-05", + "value": "Customer responsible for PII redaction; moderation API available" + } + ], + "methodology": "Review of data protection capabilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Trust Center", + "url": "https://trust.openai.com/", + "date": "2026-03-05", + "value": "SOC 2 Type II, ISO 27001, GDPR compliant" + } + ], + "methodology": "Verification of compliance certifications", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-03-05", + "value": "Zero-data-retention options available for enterprise and qualifying API customers" + } + ], + "methodology": "Enterprise feature review", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard OpenAI enterprise posture: SOC 2, no API-data training by default, zero-data-retention options. Not HIPAA eligible." + }, + + "trust_transparency": { + "overall_score": 89, + "criteria": { + "explainability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "GPT-5.4 Thinking Variant", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "Thinking variant exposes reasoning summaries; computer-use actions are step-logged" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "~33% fewer factual errors than GPT-5.2" + } + ], + "methodology": "Factual accuracy testing on QA datasets", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-03-05", + "value": "Regular bias testing and red-teaming program" + } + ], + "methodology": "Bias benchmarks and demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Documentation", + "url": "https://platform.openai.com/docs/models", + "date": "2026-03-05", + "value": "Better uncertainty expression accompanying the factual-error reduction" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "GPT-5.4 Announcement and System Card", + "url": "https://openai.com/index/introducing-gpt-5-4/", + "date": "2026-03-05", + "value": "Detailed release documentation covering variants, computer use, and safety evaluations" + } + ], + "methodology": "Documentation completeness and clarity review", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Blog", + "url": "https://openai.com/blog", + "date": "2026-03-05", + "value": "General description of training approach; specific sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Safety Systems", + "url": "https://openai.com/safety", + "date": "2026-03-05", + "value": "Multi-layer safety guardrails including computer-use action confirmation" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Marked transparency improvement via the ~33% factual-error reduction. Computer-use action logging aids auditability of agentic runs." + }, + + "operational_excellence": { + "overall_score": 94, + "criteria": { + "api_design_quality": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI API Reference", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-03-05", + "value": "Responses API with streaming, function calling, vision, and native computer use" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI SDKs", + "url": "https://github.com/openai", + "date": "2026-03-05", + "value": "Official SDKs for Python, Node.js, Go, .NET, actively maintained" + } + ], + "methodology": "SDK quality, documentation, and maintenance review", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-04-24", + "value": "Clear deprecation schedule; GPT-5.4 remains supported but OpenAI designates GPT-5.5 as the migration target for the GPT-5.x line" + } + ], + "methodology": "Review of versioning policy and historical deprecation practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Dashboard", + "url": "https://platform.openai.com/usage", + "date": "2026-03-05", + "value": "Detailed usage dashboard with costs, tokens, rate limits" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Support", + "url": "https://help.openai.com/", + "date": "2026-03-05", + "value": "24/7 support, comprehensive docs, active developer community" + } + ], + "methodology": "Support and documentation assessment", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Platform", + "url": "https://platform.openai.com/docs", + "date": "2026-03-05", + "value": "Largest AI ecosystem; full variant family (Thinking, Pro, mini, nano) for cost/latency tiers" + } + ], + "methodology": "Ecosystem breadth and depth analysis", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-03-05", + "value": "Standard commercial terms with usage policies" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Excellent operational maturity with the full variant family. Procurement caveat: superseded by GPT-5.5 only six weeks after release; plan migrations accordingly." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 92, + "notes": "Capable generalist coder with ~1.05M context, but GPT-5.3-Codex and GPT-5.5 lead on agentic coding benchmarks.", + "alternatives": ["gpt-5-3-codex", "gpt-5-5"] + }, + "customer-support": { + "overall": 93, + "notes": "mini and nano variants give strong cost/latency tiers; ~33% fewer factual errors improves answer reliability.", + "alternatives": ["gemini-3-5-flash", "gpt-5-5"] + }, + "content-creation": { + "overall": 93, + "notes": "Strong general writing with long-context coherence. Good value at $2.50/$15 vs GPT-5.5's $5/$30.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "data-analysis": { + "overall": 93, + "notes": "Solid analytical reasoning with ~1.05M context for whole-dataset work; GPT-5.5 leads on frontier math.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "research-assistant": { + "overall": 93, + "notes": "Reduced factual errors and very long context suit research; native computer use can drive live source gathering.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 86, + "notes": "Good document analysis but not HIPAA eligible; 30-day default retention may be a concern.", + "alternatives": ["claude-opus-4-8", "claude-fable-5"] + }, + "healthcare": { + "overall": 82, + "notes": "Not HIPAA eligible. Improved factuality helps clinical documentation but privacy controls remain a gap.", + "alternatives": ["claude-opus-4-8", "claude-fable-5"] + }, + "financial-analysis": { + "overall": 92, + "notes": "Strong quantitative reasoning at lower cost than GPT-5.5; computer use enables workflow automation across financial tools.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "education": { + "overall": 93, + "notes": "Reliable explanations with fewer factual errors; nano/mini tiers economical for high-volume tutoring.", + "alternatives": ["gpt-5-5", "gemini-3-5-flash"] + }, + "creative-writing": { + "overall": 91, + "notes": "Capable narrative writing and good instruction following; flagship successors edge it on nuance.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + } + }, + + "strengths": [ + "Native computer use: 75% OSWorld-Verified (vs GPT-5.2's 47.3%)", + "~33% fewer factual errors than GPT-5.2", + "~1.05M token input context with 128K output", + "Full variant family: Thinking, Pro, mini, nano for cost/latency tiers", + "Better price-performance than GPT-5.5 ($2.50/$15 vs $5/$30)", + "Mature OpenAI ecosystem and tooling" + ], + + "limitations": [ + "Superseded by GPT-5.5 six weeks after release — previous-generation flagship", + "Input pricing doubles above 272K input tokens", + "Not HIPAA eligible", + "30-day default API data retention", + "Behind GPT-5.5 on reasoning and coding benchmarks (e.g., 75% vs 78.7% OSWorld-Verified)", + "Computer use expands prompt-injection attack surface" + ], + + "best_for": [ + "Computer-use and desktop automation workflows", + "Cost-conscious teams wanting near-flagship quality at half GPT-5.5's price", + "Long-context workloads under 272K input tokens (before price doubling)", + "High-volume tiered deployments using mini/nano variants" + ], + + "not_recommended_for": [ + "New projects standardizing long-term (GPT-5.5 is the designated migration target)", + "HIPAA-compliant healthcare applications", + "Frontier math and abstract reasoning where GPT-5.5 leads decisively" + ], + + "metadata": { + "pricing": { + "input": "$2.50 per 1M tokens", + "output": "$15.00 per 1M tokens", + "notes": "Applies up to 272K input tokens; pricing doubles above that threshold. Thinking, Pro, mini, and nano variants priced separately.", + "last_verified": "2026-06-10" + }, + "context_window": 1050000, + "max_output": 128000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Russian", + "Arabic", + "Hindi", + "50+ languages" + ], + "modalities": ["text", "vision", "computer-use"], + "api_endpoint": "https://api.openai.com/v1/responses", + "open_source": false, + "architecture": "Transformer-based with native computer-use capability and variant family", + "parameters": "Not disclosed", + "knowledge_cutoff": "Late 2025" + }, + + "related_entities": ["gpt-5-5", "gpt-5-3-codex", "gpt-5-2", "gemini-3-1-pro"], + + "tags": [ + "computer-use", + "previous-flagship", + "million-token-context", + "factuality", + "variant-family", + "agentic", + "multimodal" + ] +} diff --git a/data/models/gpt-5-5.json b/data/models/gpt-5-5.json new file mode 100644 index 0000000..32f320d --- /dev/null +++ b/data/models/gpt-5-5.json @@ -0,0 +1,665 @@ +{ + "id": "gpt-5-5", + "type": "model", + "name": "GPT-5.5", + "provider": "OpenAI", + "version": "gpt-5-5-2026-04-24", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "OpenAI's current flagship (codename 'Spud') and first fully retrained base model since GPT-4.5. ~1.05M context, 85.0% ARC-AGI-2, 93.6% GPQA Diamond, 58.6% SWE-Bench Pro. Designated migration target for most of the GPT-5.x line.", + "website": "https://openai.com/index/introducing-gpt-5-5/", + + "trust_vector": { + "performance_reliability": { + "overall_score": 97, + "criteria": { + "task_accuracy_code": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "SWE-Bench Pro", + "url": "https://www.vellum.ai/blog/everything-you-need-to-know-about-gpt-5-5", + "date": "2026-04-23", + "value": "58.6% on SWE-Bench Pro (state-of-the-art at release)" + }, + { + "source": "Terminal-Bench 2.0", + "url": "https://www.vellum.ai/blog/everything-you-need-to-know-about-gpt-5-5", + "date": "2026-04-23", + "value": "82.7% on command-line agentic tasks" + } + ], + "methodology": "Industry-standard coding and terminal benchmarks measuring real-world software engineering tasks", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "GPQA Diamond", + "url": "https://www.vellum.ai/blog/everything-you-need-to-know-about-gpt-5-5", + "date": "2026-04-23", + "value": "93.6% (PhD-level science questions)" + }, + { + "source": "ARC-AGI-2", + "url": "https://www.vellum.ai/blog/everything-you-need-to-know-about-gpt-5-5", + "date": "2026-04-23", + "value": "85.0% (large jump over GPT-5.2's 52.9%, industry-leading abstract reasoning)" + }, + { + "source": "FrontierMath Tier 1-3", + "url": "https://www.vellum.ai/blog/everything-you-need-to-know-about-gpt-5-5", + "date": "2026-04-23", + "value": "51.7% (up from GPT-5.2's 40.3%)" + } + ], + "methodology": "PhD-level science, frontier mathematics, and abstract reasoning benchmarks", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "GDPval", + "url": "https://www.vellum.ai/blog/everything-you-need-to-know-about-gpt-5-5", + "date": "2026-04-23", + "value": "84.9% on economically valuable knowledge-work tasks" + }, + { + "source": "OSWorld-Verified", + "url": "https://www.vellum.ai/blog/everything-you-need-to-know-about-gpt-5-5", + "date": "2026-04-23", + "value": "78.7% on computer-use tasks (improving on GPT-5.4's 75%)" + } + ], + "methodology": "Expert-comparison knowledge work and computer-use benchmarks", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-5/", + "date": "2026-04-23", + "value": "First fully retrained base since GPT-4.5; ~40% fewer output tokens than GPT-5.4 for equivalent quality" + } + ], + "methodology": "Internal consistency testing reported by provider across reasoning effort levels", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "1.0s (standard) / longer with extended reasoning", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/", + "date": "2026-05-15", + "value": "Token efficiency gains (~40% fewer output tokens) reduce end-to-end response times vs GPT-5.4" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "4.0s (standard reasoning effort)", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/", + "date": "2026-05-15", + "value": "p95 latency varies significantly with reasoning effort setting" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "~1,050,000 tokens input / 128,000 output", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Documentation", + "url": "https://platform.openai.com/docs/models", + "date": "2026-04-24", + "value": "~1.05M token input context, 128K max output tokens" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 99, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Status", + "url": "https://status.openai.com/", + "date": "2026-06-01", + "value": "99.9% uptime (last 90 days)" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "First fully retrained base since GPT-4.5. State-of-the-art across reasoning (85.0% ARC-AGI-2, 93.6% GPQA) and agentic coding (82.7% Terminal-Bench 2.0). ~40% more token-efficient than GPT-5.4." + }, + + "security": { + "overall_score": 89, + "criteria": { + "prompt_injection_resistance": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Research", + "url": "https://openai.com/safety", + "date": "2026-04-23", + "value": "Strengthened injection defenses for agentic and computer-use workflows" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection attacks", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 91, + "confidence": "medium", + "evidence": [ + { + "source": "GPT-5.5 Announcement", + "url": "https://openai.com/index/introducing-gpt-5-5/", + "date": "2026-04-23", + "value": "Retrained base with updated safety training; improved refusal robustness over GPT-5.4" + } + ], + "methodology": "Adversarial prompt testing against jailbreak datasets", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Privacy Policy", + "url": "https://openai.com/policies/privacy-policy", + "date": "2026-04-24", + "value": "No training on API data by default" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-04-23", + "value": "Multi-layer safety stack carried forward and recalibrated for the retrained base" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Platform Docs", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-04-24", + "value": "API key + OAuth2 authentication, HTTPS only, rate limiting" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Mature multi-layer safety stack. Retrained base required full safety recalibration, which OpenAI reports as complete; long-tail agentic behaviors still being characterized by third parties." + }, + + "privacy_compliance": { + "overall_score": 87, + "criteria": { + "data_residency": { + "value": "US, EU", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-04-24", + "value": "Data residency options for enterprise customers" + } + ], + "methodology": "Review of enterprise documentation", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Data Controls", + "url": "https://openai.com/policies/usage-policies", + "date": "2026-04-24", + "value": "API data not used for training by default" + } + ], + "methodology": "Policy review of data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "30 days (zero retention available)", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-04-24", + "value": "30-day default API log retention; zero-data-retention options for qualifying customers" + } + ], + "methodology": "Terms of service and enterprise documentation review", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Tools", + "url": "https://platform.openai.com/docs/guides/safety", + "date": "2026-04-24", + "value": "Customer responsible for PII redaction; moderation API available" + } + ], + "methodology": "Review of data protection capabilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Trust Center", + "url": "https://trust.openai.com/", + "date": "2026-04-24", + "value": "SOC 2 Type II, ISO 27001, GDPR compliant" + } + ], + "methodology": "Verification of compliance certifications", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-04-24", + "value": "Zero-data-retention options available for enterprise and qualifying API customers" + } + ], + "methodology": "Enterprise feature review", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard OpenAI enterprise posture: SOC 2, no API-data training by default, 30-day default retention with zero-data-retention options." + }, + + "trust_transparency": { + "overall_score": 90, + "criteria": { + "explainability": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "GPT-5.5 Announcement", + "url": "https://openai.com/index/introducing-gpt-5-5/", + "date": "2026-04-23", + "value": "Adjustable reasoning effort with visible reasoning summaries" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Announcement", + "url": "https://openai.com/index/introducing-gpt-5-5/", + "date": "2026-04-23", + "value": "Retrained base continues factuality gains; builds on GPT-5.4's ~33% factual-error reduction" + } + ], + "methodology": "Factual accuracy testing on QA datasets", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-04-23", + "value": "Regular bias testing and red-teaming program" + } + ], + "methodology": "Bias benchmarks and demographic testing", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Documentation", + "url": "https://platform.openai.com/docs/models", + "date": "2026-04-24", + "value": "Improved calibrated uncertainty expression with reduced confident errors" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "GPT-5.5 Announcement and System Card", + "url": "https://openai.com/index/introducing-gpt-5-5/", + "date": "2026-04-23", + "value": "Detailed release documentation covering capabilities, benchmarks, and migration guidance" + } + ], + "methodology": "Documentation completeness and clarity review", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Blog", + "url": "https://openai.com/blog", + "date": "2026-04-23", + "value": "General description of training approach; specific sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Safety Systems", + "url": "https://openai.com/safety", + "date": "2026-04-23", + "value": "Multi-layer safety guardrails with agentic-workflow protections" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong transparency with reasoning summaries and detailed release documentation. Training data disclosure remains at industry-standard (limited) level." + }, + + "operational_excellence": { + "overall_score": 94, + "criteria": { + "api_design_quality": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI API Reference", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-04-24", + "value": "Responses API with streaming, function calling, vision, computer use, reasoning effort control" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI SDKs", + "url": "https://github.com/openai", + "date": "2026-04-24", + "value": "Official SDKs for Python, Node.js, Go, .NET, actively maintained" + } + ], + "methodology": "SDK quality, documentation, and maintenance review", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-04-24", + "value": "Published deprecation schedule; GPT-5.5 is the designated migration target for most of the GPT-5.x line" + } + ], + "methodology": "Review of versioning policy and historical deprecation practices", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Dashboard", + "url": "https://platform.openai.com/usage", + "date": "2026-04-24", + "value": "Detailed usage dashboard with costs, tokens, rate limits" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Support", + "url": "https://help.openai.com/", + "date": "2026-04-24", + "value": "24/7 support, comprehensive docs, active developer community" + } + ], + "methodology": "Support and documentation assessment", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Platform", + "url": "https://platform.openai.com/docs", + "date": "2026-04-24", + "value": "Largest AI ecosystem; available in ChatGPT (2026-04-23) and API (2026-04-24) with Batch and Flex tiers" + } + ], + "methodology": "Ecosystem breadth and depth analysis", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-04-24", + "value": "Standard commercial terms with usage policies" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Industry-leading operational maturity. As the designated GPT-5.x migration target, GPT-5.5 offers the longest expected support horizon in the OpenAI lineup." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 96, + "notes": "58.6% SWE-Bench Pro and 82.7% Terminal-Bench 2.0. ~1.05M context fits very large codebases. Codex variants remain preferable for dedicated agentic coding pipelines.", + "alternatives": ["gpt-5-3-codex", "claude-opus-4-8"] + }, + "customer-support": { + "overall": 93, + "notes": "Token efficiency (~40% fewer output tokens) lowers cost per conversation. $5/$30 pricing is premium for high-volume support.", + "alternatives": ["gpt-5-4", "gemini-3-5-flash"] + }, + "content-creation": { + "overall": 94, + "notes": "Retrained base produces concise, higher-quality drafts. Strong long-form coherence over very long contexts.", + "alternatives": ["claude-opus-4-8", "gpt-5-4"] + }, + "data-analysis": { + "overall": 96, + "notes": "93.6% GPQA and 51.7% FrontierMath T1-3 support rigorous quantitative work. ~1.05M context enables whole-dataset reasoning.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "research-assistant": { + "overall": 96, + "notes": "84.9% GDPval on expert knowledge work with ~1.05M context for literature-scale inputs.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "legal-compliance": { + "overall": 87, + "notes": "Strong document reasoning; SOC 2 and zero-data-retention options available, but not HIPAA eligible and 30-day default retention.", + "alternatives": ["claude-opus-4-8", "claude-fable-5"] + }, + "healthcare": { + "overall": 83, + "notes": "Excellent clinical reasoning (93.6% GPQA) but not HIPAA eligible; privacy controls less strict than Anthropic's.", + "alternatives": ["claude-opus-4-8", "claude-fable-5"] + }, + "financial-analysis": { + "overall": 95, + "notes": "Frontier math performance (51.7% FrontierMath T1-3) and GDPval results support complex financial modeling.", + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] + }, + "education": { + "overall": 95, + "notes": "Top-tier STEM reasoning with adjustable effort for tutoring at different depths. Reduced hallucinations vs prior generations.", + "alternatives": ["gpt-5-4", "gemini-3-1-pro"] + }, + "creative-writing": { + "overall": 92, + "notes": "Strong narrative quality; conciseness bias from token-efficiency training can need prompting for expansive prose.", + "alternatives": ["claude-opus-4-8", "gpt-5-4"] + } + }, + + "strengths": [ + "Industry-leading abstract reasoning: 85.0% ARC-AGI-2, 93.6% GPQA Diamond", + "State-of-the-art agentic coding: 58.6% SWE-Bench Pro, 82.7% Terminal-Bench 2.0", + "~1.05M token input context with 128K output", + "First fully retrained base since GPT-4.5 with ~40% fewer output tokens than GPT-5.4", + "84.9% GDPval on economically valuable knowledge work", + "Designated long-term migration target for the GPT-5.x line", + "Batch/Flex tiers at 50% discount" + ], + + "limitations": [ + "Premium pricing: $5/$30 per 1M tokens (2x GPT-5.4's base rate)", + "Not HIPAA eligible", + "30-day default API data retention (zero retention requires enterprise arrangement)", + "GPT-5.5 Pro is very expensive ($30/$180 per 1M)", + "Recently retrained base — long-tail behaviors less battle-tested than GPT-5.x predecessors", + "Training data transparency limited (industry standard)" + ], + + "best_for": [ + "Frontier reasoning and research tasks (ARC-AGI-2, GPQA, FrontierMath leader)", + "Large-codebase software engineering with ~1.05M context", + "Agentic and computer-use workflows (78.7% OSWorld-Verified)", + "Teams consolidating from older GPT-5.x models onto a long-support flagship", + "Expert-level knowledge work (84.9% GDPval)" + ], + + "not_recommended_for": [ + "HIPAA-compliant healthcare applications", + "Cost-sensitive high-volume inference (use mini/nano tiers or Batch)", + "Applications requiring zero data retention without enterprise agreements" + ], + + "metadata": { + "pricing": { + "input": "$5.00 per 1M tokens", + "output": "$30.00 per 1M tokens", + "notes": "Batch and Flex processing at 50% discount. GPT-5.5 Pro priced at $30/$180 per 1M tokens.", + "last_verified": "2026-06-10" + }, + "context_window": 1050000, + "max_output": 128000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Russian", + "Arabic", + "Hindi", + "50+ languages" + ], + "modalities": ["text", "vision", "computer-use"], + "api_endpoint": "https://api.openai.com/v1/responses", + "open_source": false, + "architecture": "Transformer-based; first fully retrained base since GPT-4.5 ('Spud')", + "parameters": "Not disclosed", + "knowledge_cutoff": "December 2025" + }, + + "related_entities": ["gpt-5-4", "gpt-5-3-codex", "gpt-5-2", "claude-opus-4-8", "gemini-3-1-pro"], + + "tags": [ + "reasoning", + "flagship", + "million-token-context", + "agentic", + "computer-use", + "retrained-base", + "migration-target", + "token-efficient" + ] +} diff --git a/data/models/grok-3-beta.json b/data/models/grok-3-beta.json index bc25534..d13263d 100644 --- a/data/models/grok-3-beta.json +++ b/data/models/grok-3-beta.json @@ -4,9 +4,9 @@ "name": "Grok 3 [Beta]", "provider": "xAI", "version": "Beta", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "xAI's flagship Grok 3 model in beta, featuring exceptional coding performance and real-time knowledge integration via X platform. Designed for cutting-edge applications requiring both high accuracy and current information.", + "description": "RETIRED: xAI retired Grok 3 on 2026-05-15; retired API slugs now silently redirect to Grok 4.3 at Grok 4.3 pricing. Historically xAI's flagship beta model with exceptional coding performance and real-time knowledge via X platform. Migrate to Grok 4.3 (current xAI flagship) or Grok 4.1.", "website": "https://x.ai/grok", "trust_vector": { "performance_reliability": { @@ -421,7 +421,7 @@ "notes": "Good transparency for beta product. Real-time X integration provides current information. Some aspects still evolving." }, "operational_excellence": { - "overall_score": 81, + "overall_score": 78, "criteria": { "api_design_quality": { "score": 84, @@ -453,14 +453,20 @@ "notes": "SDKs still maturing" }, "versioning_policy": { - "score": 78, - "confidence": "medium", + "score": 68, + "confidence": "high", "evidence": [ { "source": "xAI API Versioning", "url": "https://docs.x.ai/versioning", "date": "2025-01-20", "value": "Beta versioning approach" + }, + { + "source": "xAI May 15 Retirement Migration Guide", + "url": "https://docs.x.ai/developers/migration/may-15-retirement", + "date": "2026-06-10", + "value": "grok-3 retired 2026-05-15; retired slugs silently redirect to grok-4.3 at grok-4.3 pricing" } ], "methodology": "Review of versioning", @@ -495,7 +501,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 76, + "score": 66, "confidence": "medium", "evidence": [ { @@ -523,7 +529,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Good operational foundation for beta product. Ecosystem and tooling still maturing." + "notes": "Model retired 2026-05-15; retired slugs silently redirect to grok-4.3 at grok-4.3 pricing. Versioning and ecosystem scores reduced to reflect retirement." } }, "use_case_ratings": { @@ -622,7 +628,8 @@ "Limited ecosystem maturity compared to established models", "30-day data retention period", "Not HIPAA eligible", - "Support and documentation still developing" + "Support and documentation still developing", + "RETIRED 2026-05-15: xAI no longer serves grok-3; retired slugs silently redirect to grok-4.3 at grok-4.3 pricing" ], "best_for": [ "Cutting-edge software development requiring best-in-class coding", @@ -669,12 +676,11 @@ "parameters": "Not disclosed (large-scale)" }, "related_entities": [ - "openai-o3", - "claude-sonnet-4-5", - "gpt-4-1", - "llama-4-behemoth" + "grok-4-3", + "grok-4-1" ], "tags": [ + "retired", "beta", "coding", "real-time", diff --git a/data/models/grok-4-1.json b/data/models/grok-4-1.json new file mode 100644 index 0000000..21c6081 --- /dev/null +++ b/data/models/grok-4-1.json @@ -0,0 +1,633 @@ +{ + "id": "grok-4-1", + "type": "model", + "name": "Grok 4.1", + "provider": "xAI", + "version": "4.1 (2025-11-17)", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "xAI's late-2025 flagship that debuted #1 on LMArena Text (1483 Elo) and led EQ-Bench3 for emotional intelligence, with a 2M token context window. Now superseded by Grok 4.3; the grok-4-1-fast variants were retired on 2026-05-15.", + "website": "https://x.ai/news", + "trust_vector": { + "performance_reliability": { + "overall_score": 92, + "criteria": { + "task_accuracy_code": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "llm-stats", + "url": "https://llm-stats.com/models/grok-4.1-2025-11-17", + "date": "2025-11-17", + "value": "Strong coding performance, competitive with late-2025 frontier peers" + } + ], + "methodology": "Review of third-party benchmark aggregator data", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "Substantial reasoning gains over Grok 4 with reduced hallucination rate" + }, + { + "source": "EQ-Bench3", + "url": "https://eqbench.com/", + "date": "2025-11-17", + "value": "Leader on EQ-Bench3 emotional intelligence benchmark at launch" + } + ], + "methodology": "Provider launch evaluations and independent benchmark leaderboards", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "LMArena Text Leaderboard", + "url": "https://lmarena.ai/leaderboard", + "date": "2025-11-17", + "value": "#1 on LMArena Text at launch with 1483 Elo" + }, + { + "source": "llm-stats", + "url": "https://llm-stats.com/models/grok-4.1-2025-11-17", + "date": "2025-11-17", + "value": "Top-tier general knowledge and conversational quality" + } + ], + "methodology": "Crowdsourced arena comparisons and aggregator metrics", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "Reduced hallucination and improved instruction adherence vs Grok 4" + } + ], + "methodology": "Review of provider claims and community repeated-prompt reports", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "~2.5s (standard); sub-second with 4.1 Fast", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2025-12-01", + "value": "Standard 4.1 ~2.5s typical; Fast variant optimized for low-latency agentic use" + } + ], + "methodology": "Median latency from third-party API benchmarking", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "~6.0s (standard)", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2025-12-01", + "value": "Tail latency higher with extended reasoning engaged" + } + ], + "methodology": "95th percentile response time from third-party benchmarking", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "2,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "llm-stats", + "url": "https://llm-stats.com/models/grok-4.1-2025-11-17", + "date": "2025-11-17", + "value": "2M token context window" + } + ], + "methodology": "Official specification reflected in aggregator listings", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Status Page", + "url": "https://status.x.ai/", + "date": "2026-05-01", + "value": "Stable availability through its lifecycle; Fast variants retired 2026-05-15" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Released 2025-11-17 and #1 on LMArena Text at launch (1483 Elo) with EQ-Bench3 leadership. Superseded by Grok 4.3 as xAI's flagship; grok-4-1-fast variants retired 2026-05-15." + }, + "security": { + "overall_score": 83, + "criteria": { + "prompt_injection_resistance": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "Improved system prompt adherence; limited published red-team data" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection patterns", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "Safety tuning improvements cited in 4.1 release notes" + } + ], + "methodology": "Review of adversarial prompt testing and community reports", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2025-11-01", + "value": "API data handling documented; fewer contractual controls than major enterprise providers" + } + ], + "methodology": "Analysis of privacy policies and data handling commitments", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "Lower hallucination rate and improved refusal calibration vs Grok 4" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI API Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "API key authentication, HTTPS only, rate limiting" + } + ], + "methodology": "Review of API security features", + "last_verified": "2026-06-10" + } + }, + "notes": "Solid baseline; xAI publishes less safety evaluation detail than Anthropic, OpenAI, or Google." + }, + "privacy_compliance": { + "overall_score": 76, + "criteria": { + "data_residency": { + "value": "US (primary)", + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "US-based infrastructure; no published regional residency options" + } + ], + "methodology": "Review of provider documentation", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2025-11-01", + "value": "API customer data not used for training by default per policy" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "30 days (standard API)", + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2025-11-01", + "value": "Limited retention for abuse monitoring; zero-retention via enterprise agreement" + } + ], + "methodology": "Review of terms and retention policies", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "Customer responsible for PII redaction" + } + ], + "methodology": "Review of data protection capabilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Trust Center", + "url": "https://trust.x.ai/", + "date": "2025-12-01", + "value": "SOC 2 Type II; no HIPAA BAA program, fewer attestations than major providers" + } + ], + "methodology": "Verification of compliance certifications", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Trust Center", + "url": "https://trust.x.ai/", + "date": "2025-12-01", + "value": "Zero-data-retention only via negotiated enterprise terms" + } + ], + "methodology": "Review of data handling practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Same thinner-than-peers xAI compliance posture as the rest of the Grok line: SOC 2 but no HIPAA program." + }, + "trust_transparency": { + "overall_score": 82, + "criteria": { + "explainability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "Reasoning traces available; improved explanation quality vs Grok 4" + } + ], + "methodology": "Evaluation of reasoning transparency", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "Headline launch claim: significantly reduced hallucination rate vs Grok 4" + } + ], + "methodology": "Review of provider factuality evaluations and community testing", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 76, + "confidence": "low", + "evidence": [ + { + "source": "xAI Public Statements", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "Limited published bias evaluation detail" + } + ], + "methodology": "Review of bias disclosures and independent reporting", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "Reasonable uncertainty expression, improved with 4.1 tuning" + } + ], + "methodology": "Qualitative assessment of confidence expression", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "Model documentation with capabilities and pricing; less depth than peers' system cards" + } + ], + "methodology": "Review of documentation completeness", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Public Statements", + "url": "https://x.ai/news", + "date": "2025-11-17", + "value": "General description including X platform data; detailed sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "Built-in moderation with lighter-touch defaults than peers" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Notable for launch emphasis on hallucination reduction and emotional intelligence (EQ-Bench3 leader)." + }, + "operational_excellence": { + "overall_score": 81, + "criteria": { + "api_design_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "xAI API Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "OpenAI-compatible API; Fast variant tailored for agentic tool-calling" + } + ], + "methodology": "Review of API design and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI SDKs", + "url": "https://github.com/xai-org", + "date": "2025-11-17", + "value": "Official SDKs plus OpenAI client compatibility" + } + ], + "methodology": "Review of SDK quality and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "xAI Migration Guide (May 15 Retirement)", + "url": "https://docs.x.ai/developers/migration/may-15-retirement", + "date": "2026-05-15", + "value": "grok-4-1-fast variants retired 2026-05-15, about six months after launch, with retired slugs redirecting to grok-4.3" + } + ], + "methodology": "Review of deprecation timeline; rapid retirement and silent redirection penalize lifecycle predictability", + "last_verified": "2026-06-10", + "notes": "Six-month lifespan for Fast variants is short for production planning" + }, + "monitoring_observability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Console", + "url": "https://console.x.ai/", + "date": "2025-11-17", + "value": "Usage dashboard with spend and rate limit visibility" + } + ], + "methodology": "Review of monitoring tools", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2025-11-17", + "value": "Improving documentation; support channels lighter than major cloud providers" + } + ], + "methodology": "Assessment of documentation and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "llm-stats", + "url": "https://llm-stats.com/models/grok-4.1-2025-11-17", + "date": "2025-11-17", + "value": "Broad availability via aggregators and frameworks during its flagship period" + } + ], + "methodology": "Analysis of third-party integrations", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "xAI Terms of Service", + "url": "https://x.ai/legal/terms-of-service", + "date": "2025-11-01", + "value": "Clear commercial API terms" + } + ], + "methodology": "Review of licensing terms", + "last_verified": "2026-06-10" + } + }, + "notes": "Solid operations during its run, but the 2026-05-15 retirement of Fast variants and supersession by Grok 4.3 make this a legacy choice for new builds." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 89, + "notes": "Strong coding for its generation, but Grok 4.3 supersedes it at far lower cost.", + "alternatives": ["grok-4-3", "claude-opus-4-8"] + }, + "customer-support": { + "overall": 91, + "notes": "EQ-Bench3 leadership translates to excellent empathetic support conversations.", + "alternatives": ["claude-sonnet-4-6", "grok-4-3"] + }, + "content-creation": { + "overall": 90, + "notes": "Top-rated conversational and writing quality at launch (#1 LMArena Text).", + "alternatives": ["gpt-5-5", "grok-4-3"] + }, + "data-analysis": { + "overall": 88, + "notes": "2M context handles very large datasets; standard pricing ($3/$15) is high vs Grok 4.3.", + "alternatives": ["gemini-3-1-pro", "grok-4-3"] + }, + "research-assistant": { + "overall": 90, + "notes": "2M context window is among the largest available; strong synthesis quality.", + "alternatives": ["gemini-3-1-pro", "grok-4-3"] + }, + "legal-compliance": { + "overall": 74, + "notes": "Thin compliance certifications and legacy lifecycle status argue against new regulated deployments.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "healthcare": { + "overall": 68, + "notes": "No HIPAA eligibility; superseded model. Not recommended for PHI workloads.", + "alternatives": ["claude-opus-4-8", "nova-2-lite"] + }, + "financial-analysis": { + "overall": 87, + "notes": "Strong reasoning over long documents; consider lifecycle risk for production systems.", + "alternatives": ["gpt-5-5", "grok-4-3"] + }, + "education": { + "overall": 89, + "notes": "Empathetic, patient explanations backed by EQ-Bench3 leadership.", + "alternatives": ["claude-sonnet-4-6", "gemini-3-5-flash"] + }, + "creative-writing": { + "overall": 91, + "notes": "One of the strongest creative/conversational models of late 2025.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + } + }, + "strengths": [ + "#1 LMArena Text at launch (1483 Elo)", + "EQ-Bench3 leader: best-in-class emotional intelligence at release", + "2M token context window, among the largest available", + "Significantly reduced hallucination rate vs Grok 4", + "Fast variant offered very low-cost agentic inference ($0.20/$0.50 per 1M)" + ], + "limitations": [ + "Superseded by Grok 4.3 as xAI's flagship", + "grok-4-1-fast variants retired 2026-05-15 (about six months after launch)", + "Standard pricing (~$3/$15 per 1M) far above Grok 4.3's $1.25/$2.50", + "Thin enterprise compliance posture; no HIPAA eligibility", + "Retired slugs silently redirect, complicating pinned deployments" + ], + "best_for": [ + "Existing integrations not yet migrated to Grok 4.3", + "Empathetic conversational experiences leveraging its EQ strengths", + "Ultra-long-context workloads needing the 2M token window", + "Historical benchmarking and model comparisons" + ], + "not_recommended_for": [ + "New production builds (superseded; Fast variants already retired)", + "Healthcare or regulated workloads requiring compliance attestations", + "Cost-sensitive workloads (Grok 4.3 is cheaper and newer)", + "Teams requiring long, predictable model lifecycles" + ], + "metadata": { + "pricing": { + "input": "$3.00 per 1M tokens", + "output": "$15.00 per 1M tokens", + "notes": "Grok 4.1 Fast variant was $0.20/$0.50 per 1M tokens before its 2026-05-15 retirement. Standard 4.1 superseded by Grok 4.3 ($1.25/$2.50).", + "last_verified": "2026-06-10" + }, + "context_window": 2000000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic" + ], + "modalities": ["text", "image (input)"], + "api_endpoint": "https://api.x.ai/v1/chat/completions", + "open_source": false, + "architecture": "Transformer-based with reasoning and agentic tool-calling (Fast variant)", + "parameters": "Not disclosed", + "release_date": "2025-11-17", + "lifecycle_status": "Superseded by Grok 4.3; grok-4-1-fast retired 2026-05-15" + }, + "related_entities": ["grok-4-3", "grok-3-beta", "gpt-5-4", "claude-sonnet-4-6", "gemini-3-1-pro"], + "tags": [ + "superseded", + "long-context", + "emotional-intelligence", + "lmarena-leader", + "reasoning", + "legacy" + ] +} diff --git a/data/models/grok-4-3.json b/data/models/grok-4-3.json new file mode 100644 index 0000000..aafa6a8 --- /dev/null +++ b/data/models/grok-4-3.json @@ -0,0 +1,636 @@ +{ + "id": "grok-4-3", + "type": "model", + "name": "Grok 4.3", + "provider": "xAI", + "version": "4.3", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "xAI's current flagship model released in early May 2026, with a 1M token context window, reasoning, function calling, and structured outputs at aggressive pricing ($1.25/$2.50 per 1M tokens). Strong frontier performance, but a thinner enterprise compliance posture than Anthropic, OpenAI, or Google.", + "website": "https://docs.x.ai/developers/models/grok-4.3", + "trust_vector": { + "performance_reliability": { + "overall_score": 94, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Frontier coding performance positioned as flagship successor to Grok 4.1" + }, + { + "source": "llm-stats", + "url": "https://llm-stats.com/models/grok-4.3", + "date": "2026-05-06", + "value": "Competitive with frontier peers on agentic coding evaluations" + } + ], + "methodology": "Review of provider documentation and third-party benchmark aggregators", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 94, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Native reasoning mode with strong math and science performance" + } + ], + "methodology": "Review of reasoning benchmark results from provider and aggregators", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "LMArena Leaderboard", + "url": "https://lmarena.ai/leaderboard", + "date": "2026-06-01", + "value": "Top-tier placement among frontier models in crowdsourced comparisons" + }, + { + "source": "OpenRouter Model Listing", + "url": "https://openrouter.ai/x-ai/grok-4.3", + "date": "2026-04-30", + "value": "High usage and quality ratings since launch" + } + ], + "methodology": "Crowdsourced arena comparisons and aggregator quality metrics", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Structured outputs and function calling support deterministic integration patterns" + } + ], + "methodology": "Review of structured output features and community reports of repeated-prompt behavior", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "~2.0s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-01", + "value": "Typical time-to-full-response around 2s for standard prompts (non-reasoning mode)" + } + ], + "methodology": "Median latency from third-party API benchmarking", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "~5.0s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-06-01", + "value": "Tail latency higher when extended reasoning is engaged" + } + ], + "methodology": "95th percentile response time from third-party benchmarking; reasoning mode adds variance", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "xAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "1M token context window; higher per-token rate applies above 200K tokens" + } + ], + "methodology": "Official specification from provider documentation", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 96, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Status Page", + "url": "https://status.x.ai/", + "date": "2026-06-01", + "value": "Generally stable availability since launch with occasional incidents" + } + ], + "methodology": "Historical uptime data from official status page", + "last_verified": "2026-06-10" + } + }, + "notes": "Frontier-class performance with a 1M context window and reasoning, function calling, and structured outputs. Release date sources conflict (2026-04-30 per OpenRouter vs 2026-05-06 per llm-stats); xAI documentation is treated as primary." + }, + "security": { + "overall_score": 83, + "criteria": { + "prompt_injection_resistance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Hardened system prompt handling; limited published red-team data" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection patterns and review of published safety material", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2026-05-06", + "value": "Safety improvements cited at launch; less third-party adversarial testing than peers" + } + ], + "methodology": "Review of adversarial prompt testing results and community jailbreak reports", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2026-05-01", + "value": "API data handling documented; fewer contractual controls than major enterprise providers" + } + ], + "methodology": "Analysis of privacy policies and data handling commitments", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Content moderation in place; xAI publishes less safety evaluation detail than Anthropic/OpenAI/Google" + } + ], + "methodology": "Safety testing across harmful content categories and review of published evaluations", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI API Documentation", + "url": "https://docs.x.ai/", + "date": "2026-05-06", + "value": "API key authentication, HTTPS only, rate limiting, team management in console" + } + ], + "methodology": "Review of API security features and authentication mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Reasonable baseline security, but xAI publishes substantially less safety and red-team documentation than Anthropic, OpenAI, or Google." + }, + "privacy_compliance": { + "overall_score": 76, + "criteria": { + "data_residency": { + "value": "US (primary)", + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2026-05-06", + "value": "US-based infrastructure; no published regional residency options" + } + ], + "methodology": "Review of provider documentation and enterprise materials", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2026-05-01", + "value": "API customer data not used for training by default per policy" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "30 days (standard API)", + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2026-05-01", + "value": "Limited retention for abuse monitoring; zero-retention requires enterprise agreement" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2026-05-06", + "value": "Customer responsible for PII redaction; no built-in PII tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Trust Center", + "url": "https://trust.x.ai/", + "date": "2026-06-01", + "value": "SOC 2 Type II; thinner certification portfolio (no HIPAA BAA program, limited GDPR tooling) vs Anthropic/OpenAI/Google" + } + ], + "methodology": "Verification of compliance certifications against major enterprise provider baselines", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Trust Center", + "url": "https://trust.x.ai/", + "date": "2026-06-01", + "value": "Zero-data-retention available only via negotiated enterprise terms" + } + ], + "methodology": "Review of data handling practices and enterprise contract options", + "last_verified": "2026-06-10" + } + }, + "notes": "xAI's enterprise compliance posture remains thinner than Anthropic, OpenAI, or Google: SOC 2 in place but no HIPAA eligibility program and fewer regulated-industry attestations." + }, + "trust_transparency": { + "overall_score": 82, + "criteria": { + "explainability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Reasoning traces available via API for inspection" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "xAI News", + "url": "https://x.ai/news", + "date": "2026-05-06", + "value": "Continued hallucination reductions claimed at launch, building on Grok 4.1 improvements" + } + ], + "methodology": "Review of provider claims and factual QA testing", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 76, + "confidence": "low", + "evidence": [ + { + "source": "xAI Public Statements", + "url": "https://x.ai/news", + "date": "2026-05-06", + "value": "Limited published bias evaluation; past Grok versions drew scrutiny over politically tuned behavior" + } + ], + "methodology": "Review of bias benchmark disclosures and independent reporting", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Reasoning mode expresses uncertainty reasonably well" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "xAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.3", + "date": "2026-05-06", + "value": "Detailed model page with capabilities, pricing, limits, and feature support" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Public Statements", + "url": "https://x.ai/news", + "date": "2026-05-06", + "value": "General description including X platform data; detailed sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2026-05-06", + "value": "Built-in moderation with developer controls; lighter-touch defaults than peers" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Good developer-facing documentation and inspectable reasoning, but less published safety/bias evaluation than major competitors." + }, + "operational_excellence": { + "overall_score": 82, + "criteria": { + "api_design_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "xAI API Documentation", + "url": "https://docs.x.ai/", + "date": "2026-05-06", + "value": "OpenAI-compatible API with reasoning, function calling, structured outputs, and prompt caching" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI SDKs", + "url": "https://github.com/xai-org", + "date": "2026-05-06", + "value": "Official SDKs plus broad compatibility with OpenAI client libraries" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 75, + "confidence": "high", + "evidence": [ + { + "source": "xAI Migration Guide (May 15 Retirement)", + "url": "https://docs.x.ai/developers/migration/may-15-retirement", + "date": "2026-05-15", + "value": "Retired Grok model slugs silently redirect to grok-4.3 rather than returning errors" + } + ], + "methodology": "Review of deprecation/migration practices; silent redirects of retired slugs reduce predictability for pinned workloads", + "last_verified": "2026-06-10", + "notes": "Silent redirection of retired model slugs to grok-4.3 can change behavior of production systems without explicit failure signals" + }, + "monitoring_observability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Console", + "url": "https://console.x.ai/", + "date": "2026-05-06", + "value": "Usage dashboard with spend and rate limit visibility" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Documentation", + "url": "https://docs.x.ai/", + "date": "2026-05-06", + "value": "Improving documentation; support channels lighter than major cloud providers" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenRouter Model Listing", + "url": "https://openrouter.ai/x-ai/grok-4.3", + "date": "2026-04-30", + "value": "Available via OpenRouter and major LLM frameworks; growing third-party adoption" + } + ], + "methodology": "Analysis of third-party integrations and tools", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "xAI Terms of Service", + "url": "https://x.ai/legal/terms-of-service", + "date": "2026-05-01", + "value": "Clear commercial API terms; enterprise agreements available" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong API and pricing, but the May 2026 retirement wave (with silent slug redirects to grok-4.3) highlights an aggressive deprecation culture enterprises should plan around." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 94, + "notes": "Frontier-class coding with function calling and structured outputs at very competitive pricing.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "customer-support": { + "overall": 87, + "notes": "Fast, capable, and cheap for support workloads; compliance posture may limit regulated deployments.", + "alternatives": ["claude-sonnet-4-6", "gemini-3-5-flash"] + }, + "content-creation": { + "overall": 89, + "notes": "Strong long-form generation with current-events awareness from the X ecosystem.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "data-analysis": { + "overall": 91, + "notes": "Strong reasoning over large inputs; 1M context handles big datasets, with higher per-token rates above 200K.", + "alternatives": ["gemini-3-1-pro", "gpt-5-5"] + }, + "research-assistant": { + "overall": 92, + "notes": "1M context plus reasoning makes it well suited to literature-scale synthesis at low cost.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "legal-compliance": { + "overall": 76, + "notes": "Capable analytically, but thinner compliance certifications than Anthropic/OpenAI/Google providers.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "healthcare": { + "overall": 70, + "notes": "No HIPAA eligibility program; not recommended for PHI workloads.", + "alternatives": ["claude-opus-4-8", "nova-2-lite"] + }, + "financial-analysis": { + "overall": 89, + "notes": "Strong quantitative reasoning and real-time information; verify compliance requirements first.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "education": { + "overall": 89, + "notes": "Strong explanations at low cost; content controls are lighter-touch than peers.", + "alternatives": ["claude-sonnet-4-6", "gemini-3-5-flash"] + }, + "creative-writing": { + "overall": 88, + "notes": "Distinctive voice and strong creative range; fewer content restrictions than competitors.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + } + }, + "strengths": [ + "Aggressive pricing: $1.25/$2.50 per 1M tokens with $0.20 cached input", + "1M token context window", + "Full agentic feature set: reasoning, function calling, structured outputs", + "Text and image input support", + "Frontier-class performance across coding, reasoning, and general tasks", + "OpenAI-compatible API simplifies migration" + ], + "limitations": [ + "Thinner enterprise compliance posture than Anthropic, OpenAI, or Google (no HIPAA program)", + "Retired Grok model slugs silently redirect to grok-4.3, risking unannounced behavior changes", + "Higher per-token rate applies above 200K context", + "Limited published safety, bias, and red-team evaluation detail", + "Zero-data-retention only via negotiated enterprise terms", + "Conflicting release-date records across aggregators reflect lighter release documentation" + ], + "best_for": [ + "Cost-sensitive agentic and coding workloads needing frontier quality", + "Long-context analysis and research up to 1M tokens", + "Applications wanting real-time/current-events awareness", + "Teams already on OpenAI-compatible tooling seeking cheaper frontier capacity" + ], + "not_recommended_for": [ + "Healthcare workloads involving PHI (no HIPAA eligibility)", + "Regulated industries requiring deep compliance attestations", + "Production systems that cannot tolerate silent model-slug redirection", + "Organizations requiring contractual zero data retention by default" + ], + "metadata": { + "pricing": { + "input": "$1.25 per 1M tokens", + "output": "$2.50 per 1M tokens", + "notes": "Cached input $0.20 per 1M tokens. Higher per-token rate applies for requests above 200K context.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)"], + "api_endpoint": "https://api.x.ai/v1/chat/completions", + "open_source": false, + "architecture": "Transformer-based with native reasoning, function calling, and structured outputs", + "parameters": "Not disclosed", + "release_date": "Early May 2026 (2026-04-30 per OpenRouter; 2026-05-06 per llm-stats)" + }, + "related_entities": ["grok-4-1", "grok-3-beta", "gpt-5-5", "claude-opus-4-8", "gemini-3-1-pro"], + "tags": [ + "flagship", + "reasoning", + "long-context", + "function-calling", + "structured-outputs", + "cost-effective", + "real-time" + ] +} diff --git a/data/models/kimi-k2-6.json b/data/models/kimi-k2-6.json new file mode 100644 index 0000000..78ca86e --- /dev/null +++ b/data/models/kimi-k2-6.json @@ -0,0 +1,644 @@ +{ + "id": "kimi-k2-6", + "type": "model", + "name": "Kimi K2.6", + "provider": "Moonshot AI", + "version": "20260420", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Moonshot AI's open-weight 1T-parameter MoE (32B active) with vendor-reported 80.2% SWE-Bench Verified and 58.6 SWE-Bench Pro. Agent Swarm orchestration scales to 300 sub-agents and 4,000 coordinated steps for long-horizon coding.", + "website": "https://huggingface.co/moonshotai/Kimi-K2.6", + + "trust_vector": { + "performance_reliability": { + "overall_score": 91, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi K2.6 Model Card (vendor-reported)", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "SWE-Bench Verified 80.2%, SWE-Bench Pro 58.6 (vs GPT-5.4's 57.7)" + }, + { + "source": "LiveCodeBench v6 (vendor-reported)", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "89.6% on competitive programming tasks" + }, + { + "source": "MarkTechPost release coverage", + "url": "https://www.marktechpost.com/2026/04/20/moonshot-ai-releases-kimi-k2-6-with-long-horizon-coding-agent-swarm-scaling-to-300-sub-agents-and-4000-coordinated-steps/", + "date": "2026-04-20", + "value": "Long-horizon coding focus; claims open-weight state of the art on agentic coding" + } + ], + "methodology": "Vendor-reported industry-standard coding benchmarks; scores pending broad independent replication", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "Humanity's Last Exam with tools (vendor-reported)", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "54.0 on HLE-with-tools, frontier-competitive" + } + ], + "methodology": "Vendor-reported tool-augmented reasoning benchmarks requiring multi-step problem solving", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi K2.6 Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Strong general performance across knowledge benchmarks; text and vision modalities" + } + ], + "methodology": "Review of vendor benchmark suite and community evaluations across knowledge domains", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Community evaluation", + "url": "https://artificialanalysis.ai/models", + "date": "2026-05-15", + "value": "Consistent agentic behavior over long trajectories; native INT4 quantization preserves quality" + } + ], + "methodology": "Community testing of repeated runs and long-horizon agent trajectories", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "3.0s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-05-15", + "value": "Typical first-response time ~3s on first-party API; varies widely by host" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes; self-hosted latency depends on hardware", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "7.5s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-05-15", + "value": "p95 ~7.5s; long agentic chains take substantially longer by design" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "262,144 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Kimi K2.6 Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "262,144 token context window" + } + ], + "methodology": "Official specification from model card", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Platform", + "url": "https://platform.moonshot.ai/", + "date": "2026-06-01", + "value": "First-party API generally stable; open weights allow self-hosted redundancy" + } + ], + "methodology": "Review of platform availability and self-hosting fallback options", + "last_verified": "2026-06-10" + } + }, + "notes": "Vendor-reported open-weight leadership on agentic coding (80.2% SWE-Bench Verified, 58.6 SWE-Bench Pro). Agent Swarm scales to 300 sub-agents / 4,000 coordinated steps. Most headline scores are vendor-reported and await independent replication." + }, + + "security": { + "overall_score": 79, + "criteria": { + "prompt_injection_resistance": { + "score": 78, + "confidence": "low", + "evidence": [ + { + "source": "Kimi K2.6 Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Safety tuning described; no published third-party prompt-injection audit" + } + ], + "methodology": "Review of vendor safety documentation and community red-team reports against OWASP LLM01 patterns", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Community red-teaming", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-05-10", + "value": "Standard alignment tuning; open weights mean guardrails can be removed in fine-tuned derivatives" + } + ], + "methodology": "Testing against adversarial prompt datasets; open-weight deployments inherit deployer responsibility", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Privacy Policy", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-04-20", + "value": "Standard data handling on first-party API; full control when self-hosted" + } + ], + "methodology": "Analysis of privacy policies and self-hosting data-control options", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi K2.6 Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Safety post-training applied; refusal behavior comparable to other open frontier models" + } + ], + "methodology": "Safety testing across harmful content categories per vendor card and community reports", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI API Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-04-20", + "value": "API key authentication, HTTPS only, rate limiting; OpenAI-compatible endpoints" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard open-model security posture. No published third-party security audit; self-hosting shifts security responsibility to the deployer." + }, + + "privacy_compliance": { + "overall_score": 75, + "criteria": { + "data_residency": { + "value": "China (first-party API); any jurisdiction when self-hosted or via Western hosts", + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Platform Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-04-20", + "value": "Moonshot AI is a China-based provider; first-party API data processed under Chinese jurisdiction" + }, + { + "source": "OpenRouter availability", + "url": "https://openrouter.ai/models", + "date": "2026-05-01", + "value": "Available via OpenRouter and Western inference hosts, enabling non-China residency" + } + ], + "methodology": "Review of provider jurisdiction and third-party hosting options", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Privacy Policy", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-04-20", + "value": "API data usage terms standard for the segment; self-hosting removes the question entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Per Moonshot policy on first-party API (China jurisdiction); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Terms", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-04-20", + "value": "First-party retention governed by Chinese data regulations; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service and deployment-dependent retention", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-04-20", + "value": "Customer responsible for PII redaction; no managed PII tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI public materials", + "url": "https://www.moonshot.ai/", + "date": "2026-04-20", + "value": "No published SOC 2 / HIPAA / GDPR attestations for the first-party API; Western hosts may carry their own certifications" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Open weights on Hugging Face", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Self-hosting (vLLM/SGLang, native INT4) gives complete data control and zero external retention" + } + ], + "methodology": "Review of self-hosting deployment options enabling zero retention", + "last_verified": "2026-06-10" + } + }, + "notes": "First-party API operates under Chinese jurisdiction — a material caveat for Western regulated industries. Open weights fully mitigate this for organizations able to self-host or use Western inference providers." + }, + + "trust_transparency": { + "overall_score": 80, + "criteria": { + "explainability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Agent Swarm architecture", + "url": "https://www.marktechpost.com/2026/04/20/moonshot-ai-releases-kimi-k2-6-with-long-horizon-coding-agent-swarm-scaling-to-300-sub-agents-and-4000-coordinated-steps/", + "date": "2026-04-20", + "value": "Sub-agent trajectories and tool-call traces are inspectable, aiding auditability of long-horizon runs" + } + ], + "methodology": "Evaluation of reasoning and agent-trajectory transparency", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Community testing", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-05-10", + "value": "Moderate hallucination rate; tool-use grounding improves factuality in agentic mode" + } + ], + "methodology": "Testing on factual QA datasets and tool-augmented workflows", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "Kimi K2.6 Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Limited published bias evaluation" + } + ], + "methodology": "Review of published bias benchmarks and community evaluations", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Model behavior testing", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-05-10", + "value": "Expresses uncertainty adequately; no calibrated confidence outputs" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face model card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Detailed card: 1T total / 32B active MoE, 384 experts, MLA attention, native INT4, benchmarks, deployment guides" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI publications", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Architecture well documented; training data composition not disclosed in detail" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi K2.6 Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "Built-in safety tuning; deployers of open weights must layer their own guardrails" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Open weights and a detailed model card provide good architectural transparency; training data disclosure and independent benchmark verification remain limited." + }, + + "operational_excellence": { + "overall_score": 82, + "criteria": { + "api_design_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Moonshot AI API Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-04-20", + "value": "OpenAI-compatible API with streaming, tool calling, vision; Agent Swarm orchestration endpoints" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI GitHub", + "url": "https://github.com/MoonshotAI", + "date": "2026-04-20", + "value": "OpenAI-compatible so mainstream SDKs work; first-party tooling thinner than Western providers" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi release history", + "url": "https://huggingface.co/moonshotai", + "date": "2026-04-20", + "value": "K2.6 supersedes K2.5/K2; prior weights remain available, but cadence is fast" + } + ], + "methodology": "Review of versioning practices and weight availability", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Platform", + "url": "https://platform.moonshot.ai/", + "date": "2026-04-20", + "value": "Basic usage dashboard; self-hosted observability is deployer-built" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI community channels", + "url": "https://github.com/MoonshotAI", + "date": "2026-04-20", + "value": "GitHub and community support; limited English-language enterprise support" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "OpenRouter and inference ecosystem", + "url": "https://openrouter.ai/models", + "date": "2026-05-01", + "value": "Available on OpenRouter and major open-model hosts; vLLM/SGLang support with native INT4" + } + ], + "methodology": "Analysis of third-party hosting, integrations, and tooling", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Modified MIT License", + "url": "https://huggingface.co/moonshotai/Kimi-K2.6", + "date": "2026-04-20", + "value": "MIT with an attribution-UI requirement for deployments exceeding 100M MAU or $20M/month revenue" + } + ], + "methodology": "Review of licensing terms and restrictions; attribution clause is trust-relevant for large-scale commercial use", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong open-model ecosystem presence. Modified MIT license is permissive for most users but the attribution clause above 100M MAU / $20M monthly revenue requires legal review at hyperscale." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 95, + "notes": "Vendor-reported 80.2% SWE-Bench Verified and 58.6 SWE-Bench Pro; Agent Swarm excels at long-horizon multi-file engineering.", + "alternatives": ["claude-opus-4-8", "glm-5", "gpt-5-5"] + }, + "customer-support": { + "overall": 82, + "notes": "Capable but not specialized; agentic latency unnecessary for simple support flows.", + "alternatives": ["command-a-plus", "minimax-m2"] + }, + "content-creation": { + "overall": 84, + "notes": "Solid long-form generation with large context; not its differentiator.", + "alternatives": ["claude-opus-4-8", "glm-5"] + }, + "data-analysis": { + "overall": 89, + "notes": "Strong tool-augmented analysis; Agent Swarm parallelizes multi-source investigation well.", + "alternatives": ["glm-5", "deepseek-v4"] + }, + "research-assistant": { + "overall": 90, + "notes": "54.0 HLE-with-tools and 262K context make it strong for deep, tool-driven research.", + "alternatives": ["glm-5", "claude-opus-4-8"] + }, + "legal-compliance": { + "overall": 70, + "notes": "China-jurisdiction first-party API and absent Western certifications are blockers unless self-hosted.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 67, + "notes": "Not recommended via first-party API; self-hosted deployment in a compliant environment is the only viable path.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 84, + "notes": "Strong quantitative and agentic capability; data residency requires self-hosting for regulated firms.", + "alternatives": ["glm-5", "command-a-plus"] + }, + "education": { + "overall": 86, + "notes": "Strong STEM and coding tutoring at competitive pricing.", + "alternatives": ["glm-5", "deepseek-v3-2"] + }, + "creative-writing": { + "overall": 82, + "notes": "Competent creative output; optimized for agentic engineering rather than prose.", + "alternatives": ["claude-opus-4-8", "minimax-m2"] + } + }, + + "strengths": [ + "Vendor-reported open-weight leadership in agentic coding (80.2% SWE-Bench Verified, 58.6 SWE-Bench Pro vs GPT-5.4's 57.7)", + "Agent Swarm scales to 300 sub-agents and 4,000 coordinated steps for long-horizon tasks", + "Open weights with near-MIT license enable full self-hosting and data control", + "Efficient inference: 32B active of 1T total, MLA attention, native INT4 quantization", + "262,144-token context with text and vision modalities", + "Competitive API pricing (~$0.95/$4.00 per 1M tokens) and broad availability via OpenRouter" + ], + + "limitations": [ + "First-party Moonshot API processes data under Chinese jurisdiction with limited Western compliance certifications", + "Headline benchmarks are vendor-reported and await independent replication", + "Modified MIT license imposes attribution-UI requirement above 100M MAU or $20M/month revenue", + "Self-hosting a 1T-parameter MoE requires substantial GPU infrastructure even at INT4", + "Limited published bias, safety, and red-team evaluations", + "English-language enterprise support is thin compared to Western providers" + ], + + "best_for": [ + "Long-horizon autonomous coding and multi-agent engineering workflows", + "Organizations wanting frontier-level open weights they can self-host for data control", + "Tool-augmented research and deep analysis at competitive cost", + "Cost-sensitive teams needing top-tier agentic coding via OpenRouter or first-party API" + ], + + "not_recommended_for": [ + "Regulated Western workloads (healthcare, legal, finance) on the first-party API", + "Hyperscale consumer products unwilling to satisfy the attribution-UI license clause", + "Latency-critical real-time applications", + "Teams without GPU infrastructure who also cannot accept China-jurisdiction processing" + ], + + "metadata": { + "pricing": { + "input": "$0.95 per 1M tokens (approx.)", + "output": "$4.00 per 1M tokens (approx.)", + "notes": "First-party Moonshot API pricing; third-party hosts on OpenRouter vary. Self-hosting cost is infrastructure-dependent.", + "last_verified": "2026-06-10" + }, + "context_window": 262144, + "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], + "modalities": ["text", "image (input)"], + "api_endpoint": "https://api.moonshot.ai/v1/chat/completions", + "open_source": true, + "license": "Modified MIT (attribution-UI requirement above 100M MAU or $20M/month revenue)", + "architecture": "Mixture-of-Experts: 1T total / 32B active parameters, 384 experts, Multi-head Latent Attention (MLA), native INT4", + "parameters": "1T total / 32B active", + "release_date": "2026-04-20" + }, + + "related_entities": ["glm-5", "deepseek-v4", "minimax-m2", "claude-opus-4-8", "gpt-5-5"], + + "tags": [ + "coding", + "agentic", + "open-source", + "mixture-of-experts", + "long-context", + "agent-swarm", + "chinese-provider", + "self-hostable" + ] +} diff --git a/data/models/llama-3-1-405b.json b/data/models/llama-3-1-405b.json index c0e3864..8f86c78 100644 --- a/data/models/llama-3-1-405b.json +++ b/data/models/llama-3-1-405b.json @@ -4,9 +4,9 @@ "name": "Llama 3.1 405B", "provider": "Meta", "version": "llama-3.1-405b", - "last_evaluated": "2025-11-07", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Meta's largest and most capable open-source model with 405 billion parameters. Offers complete transparency, self-hosting capabilities, and competitive performance with proprietary models.", + "description": "Meta's largest open-source model with 405 billion parameters, offering complete transparency, self-hosting capabilities, and competitive performance with proprietary models. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026.", "website": "https://llama.meta.com", "trust_vector": { "performance_reliability": { @@ -533,7 +533,8 @@ "Text-only (no native vision)", "Safety guardrails can be modified (security consideration)", "Higher latency compared to smaller models", - "Complex deployment and maintenance" + "Complex deployment and maintenance", + "Legacy status: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026, so future open updates are unlikely" ], "metadata": { "license": "Llama 3.1 Community License (open for commercial use)", diff --git a/data/models/llama-3-3-70b.json b/data/models/llama-3-3-70b.json index 0ffa22a..ec40580 100644 --- a/data/models/llama-3-3-70b.json +++ b/data/models/llama-3-3-70b.json @@ -4,9 +4,9 @@ "name": "Llama 3.3 70B", "provider": "Meta", "version": "2024-12", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Meta's powerful 70B parameter Llama 3.3 model offering strong performance with open-source flexibility. Excellent balance of capability and resource efficiency for self-hosted deployments.", + "description": "Meta's powerful 70B parameter Llama 3.3 model offering strong performance with open-source flexibility and an excellent balance of capability and resource efficiency for self-hosted deployments. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026.", "website": "https://llama.meta.com/", "trust_vector": { "performance_reliability": { @@ -593,7 +593,8 @@ "Requires infrastructure for deployment", "No managed API from Meta", "Deployment expertise needed", - "Uptime depends on hosting" + "Uptime depends on hosting", + "Legacy status: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026, so future open updates are unlikely" ], "best_for": [ "Self-hosted deployments requiring data sovereignty", diff --git a/data/models/llama-4-behemoth.json b/data/models/llama-4-behemoth.json index 7a68340..54e554c 100644 --- a/data/models/llama-4-behemoth.json +++ b/data/models/llama-4-behemoth.json @@ -4,13 +4,13 @@ "name": "Llama 4 Behemoth", "provider": "Meta", "version": "2025-02", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Meta's largest and most capable open-source Llama 4 model with exceptional mathematical reasoning and knowledge. Designed for enterprises requiring state-of-the-art performance with open-source flexibility.", + "description": "Meta's announced 2T-total/288B-active parameter Llama 4 teacher model that was NEVER RELEASED. It remains 'announced, not released' as of June 2026: Meta gave no update when asked in January 2026 and has effectively exited open-weight frontier releases, shipping the proprietary closed-weight 'Muse Spark' (April 2026) instead. Scores reflect unverifiable preview-era claims; the model is not available for any deployment.", "website": "https://llama.meta.com/llama-4", "trust_vector": { "performance_reliability": { - "overall_score": 91, + "overall_score": 84, "criteria": { "task_accuracy_code": { "score": 88, @@ -129,22 +129,28 @@ "last_verified": "2025-11-08" }, "uptime": { - "score": 95, - "confidence": "medium", + "score": 60, + "confidence": "high", "evidence": [ { "source": "Self-hosted model", "url": "https://llama.meta.com/", "date": "2025-02-01", "value": "Uptime depends on hosting infrastructure" + }, + { + "source": "Wikipedia - Llama (language model)", + "url": "https://en.wikipedia.org/wiki/Llama_(language_model)", + "date": "2026-06-10", + "value": "Model was never released; weights are not available for any deployment as of June 2026" } ], "methodology": "User-controlled deployment", - "last_verified": "2025-11-08", - "notes": "Uptime dependent on deployment infrastructure" + "last_verified": "2026-06-10", + "notes": "Model never released; no deployment is possible" } }, - "notes": "Exceptional performance on mathematical reasoning (95% MATH). Strong general knowledge (73.7% MMLU). Open-source model offering enterprise-grade capabilities." + "notes": "Preview-era claims: exceptional mathematical reasoning (95% MATH) and strong general knowledge (73.7% MMLU). The model was never released, so these results cannot be independently verified." }, "security": { "overall_score": 82, @@ -419,7 +425,7 @@ "notes": "Strong transparency as open-source model. Good training data disclosure. Customizable guardrails for specific use cases." }, "operational_excellence": { - "overall_score": 84, + "overall_score": 77, "criteria": { "api_design_quality": { "score": 85, @@ -479,21 +485,28 @@ "notes": "Requires custom monitoring implementation" }, "support_quality": { - "score": 82, - "confidence": "medium", + "score": 60, + "confidence": "high", "evidence": [ { "source": "Community Support", "url": "https://github.com/meta-llama/llama4/discussions", "date": "2025-02-01", "value": "Active community, official documentation" + }, + { + "source": "SiliconANGLE", + "url": "https://siliconangle.com/2025/05/15/meta-postpone-release-llama-4-behemoth-model-report-claims/", + "date": "2026-06-10", + "value": "Release postponed in 2025; Meta provided no update when asked in January 2026" } ], "methodology": "Assessment of support channels", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10", + "notes": "No model-specific support exists since the model never shipped" }, "ecosystem_maturity": { - "score": 87, + "score": 55, "confidence": "high", "evidence": [ { @@ -501,10 +514,16 @@ "url": "https://huggingface.co/meta-llama", "date": "2025-02-01", "value": "Mature ecosystem with extensive tooling" + }, + { + "source": "Wikipedia - Llama (language model)", + "url": "https://en.wikipedia.org/wiki/Llama_(language_model)", + "date": "2026-06-10", + "value": "No ecosystem exists for Behemoth itself; the model was never released and Meta has pivoted to closed-weight models (Muse Spark, April 2026)" } ], "methodology": "Analysis of ecosystem", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10" }, "license_terms": { "score": 90, @@ -521,7 +540,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Good operational maturity with strong open-source ecosystem. Requires infrastructure expertise for deployment and monitoring." + "notes": "Operational scores are largely theoretical: the model was never released, so no deployment, support, or ecosystem exists for it. Meta shipped the closed-weight Muse Spark (April 2026) instead." } }, "use_case_ratings": { @@ -617,7 +636,8 @@ "Uptime and performance depend on hosting infrastructure", "Requires expertise to deploy and maintain", "No managed API service from Meta", - "Large model size requires substantial compute resources" + "Large model size requires substantial compute resources", + "Never released: still announced-only as of June 2026; Meta gave no update in January 2026 and pivoted to the closed-weight Muse Spark (April 2026), so weights are unavailable" ], "best_for": [ "Enterprises requiring data sovereignty and on-premise deployment", @@ -671,6 +691,7 @@ "claude-sonnet-4-5" ], "tags": [ + "unreleased", "open-source", "self-hosted", "mathematics", diff --git a/data/models/minimax-m2.json b/data/models/minimax-m2.json new file mode 100644 index 0000000..b5541e7 --- /dev/null +++ b/data/models/minimax-m2.json @@ -0,0 +1,638 @@ +{ + "id": "minimax-m2", + "type": "model", + "name": "MiniMax-M2", + "provider": "MiniMax", + "version": "20251027", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "MiniMax's MIT-licensed 230B MoE with only 10B active parameters, optimized for agentic tool calling and coding. Topped open-model agentic rankings at launch and undercut Claude pricing by roughly 92% while remaining fast due to its small active footprint.", + "website": "https://www.minimax.io/news/minimax-m2", + + "trust_vector": { + "performance_reliability": { + "overall_score": 87, + "criteria": { + "task_accuracy_code": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "MiniMax-M2 launch announcement", + "url": "https://www.minimax.io/news/minimax-m2", + "date": "2025-10-27", + "value": "Strong SWE-bench and Terminal-Bench results for an open model at launch" + }, + { + "source": "VentureBeat launch coverage", + "url": "https://venturebeat.com/ai/minimax-m2-is-the-new-king-of-open-source-llms-especially-for-agentic-tool", + "date": "2025-10-27", + "value": "Ranked the leading open-source model for agentic coding workflows at launch" + } + ], + "methodology": "Vendor benchmarks corroborated by independent press and leaderboard coverage; superseded at the top by 2026 releases", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M2 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "Competitive reasoning for its 10B-active footprint; interleaved thinking format" + } + ], + "methodology": "Vendor-reported reasoning benchmarks and community evaluation", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models", + "date": "2025-11-15", + "value": "Highest composite intelligence score among open-weight models at launch window" + } + ], + "methodology": "Independent composite benchmarking across knowledge domains", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Community evaluation", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-11-15", + "value": "Stable tool-calling behavior across long agent loops" + } + ], + "methodology": "Community testing of repeated runs and agentic trajectories", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "1.8s", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models", + "date": "2025-11-15", + "value": "Fast responses — 10B active parameters yield roughly 2x the speed of comparable dense models" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "4.0s", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2025-11-15", + "value": "p95 ~4.0s across diverse workloads" + } + ], + "methodology": "95th percentile response time across diverse workloads", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "204,800 tokens", + "confidence": "high", + "evidence": [ + { + "source": "MiniMax-M2 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "~205K token context window" + } + ], + "methodology": "Official specification from model card", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 94, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Platform", + "url": "https://platform.minimax.io/", + "date": "2026-06-01", + "value": "First-party API generally stable; open weights enable self-hosted redundancy" + } + ], + "methodology": "Review of platform availability and self-hosting fallback options", + "last_verified": "2026-06-10" + } + }, + "notes": "Was the leading open agentic model at its October 2025 launch; still strong, but 2026 releases (GLM-5, Kimi K2.6) have surpassed it on raw benchmarks. Its 10B-active design remains a standout for speed and serving cost. Successor MiniMax-M3 announced 2026-06-01 (1M context) but weights not yet published." + }, + + "security": { + "overall_score": 77, + "criteria": { + "prompt_injection_resistance": { + "score": 76, + "confidence": "low", + "evidence": [ + { + "source": "MiniMax-M2 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "Safety tuning described; no published third-party prompt-injection audit" + } + ], + "methodology": "Review of vendor documentation and community testing against OWASP LLM01 patterns", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Community red-teaming", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-11-15", + "value": "Standard alignment tuning; open weights allow guardrail removal in derivatives" + } + ], + "methodology": "Testing against adversarial prompt datasets; deployer-dependent for self-hosted use", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Privacy Policy", + "url": "https://www.minimax.io/", + "date": "2025-10-27", + "value": "Standard data handling on first-party API; full control when self-hosted" + } + ], + "methodology": "Analysis of privacy policies and self-hosting data-control options", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M2 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "Safety post-training applied; refusal behavior in line with peer open models" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax API Documentation", + "url": "https://platform.minimax.io/docs", + "date": "2025-10-27", + "value": "API key authentication, HTTPS only, rate limiting; OpenAI- and Anthropic-compatible endpoints" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Standard open-model posture without third-party audits. Self-hosting shifts security responsibility to the deployer." + }, + + "privacy_compliance": { + "overall_score": 74, + "criteria": { + "data_residency": { + "value": "China (first-party MiniMax API); any jurisdiction when self-hosted or via Western hosts", + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Platform Documentation", + "url": "https://platform.minimax.io/docs", + "date": "2025-10-27", + "value": "MiniMax is a China-based provider; first-party API data processed under Chinese jurisdiction" + }, + { + "source": "OpenRouter availability", + "url": "https://openrouter.ai/models", + "date": "2025-11-15", + "value": "MIT weights served by Western inference providers, enabling non-China residency" + } + ], + "methodology": "Review of provider jurisdiction and third-party hosting options", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Privacy Policy", + "url": "https://www.minimax.io/", + "date": "2025-10-27", + "value": "Standard API data terms; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Per MiniMax policy on first-party API (China jurisdiction); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Terms of Service", + "url": "https://www.minimax.io/", + "date": "2025-10-27", + "value": "First-party retention governed by Chinese data regulations; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service and deployment-dependent retention", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Documentation", + "url": "https://platform.minimax.io/docs", + "date": "2025-10-27", + "value": "Customer responsible for PII redaction; no managed PII tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax public materials", + "url": "https://www.minimax.io/", + "date": "2025-10-27", + "value": "No published SOC 2 / HIPAA / GDPR attestations for the first-party API" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Open weights on Hugging Face", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "MIT-licensed self-hosting gives complete data control and zero external retention; 10B active makes this unusually affordable" + } + ], + "methodology": "Review of self-hosting deployment options enabling zero retention", + "last_verified": "2026-06-10" + } + }, + "notes": "First-party MiniMax API operates under Chinese jurisdiction — a material caveat for Western regulated industries. The small 10B-active footprint makes self-hosted mitigation cheaper than for other frontier-scale open models." + }, + + "trust_transparency": { + "overall_score": 78, + "criteria": { + "explainability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M2 documentation", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "Interleaved thinking format exposes reasoning between tool calls, aiding agent-loop auditability" + } + ], + "methodology": "Evaluation of reasoning transparency and trajectory inspectability", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Community testing", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-11-15", + "value": "Moderate hallucination rate; tool-grounded workflows perform better than closed-book QA" + } + ], + "methodology": "Testing on factual QA datasets and tool-augmented workflows", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "MiniMax-M2 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "Limited published bias evaluation" + } + ], + "methodology": "Review of published bias benchmarks and community evaluations", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Model behavior testing", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-11-15", + "value": "Basic uncertainty expression; no calibrated confidence outputs" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face model card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "Clear documentation of 230B/10B MoE architecture, MIT license, benchmarks, and deployment guidance" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax publications", + "url": "https://www.minimax.io/news/minimax-m2", + "date": "2025-10-27", + "value": "Architecture documented; training data composition not disclosed in detail" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M2 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "Built-in safety tuning; deployers of open weights must layer their own guardrails" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Open weights and interleaved-thinking traces provide reasonable transparency; training data disclosure and formal bias/safety evaluations are limited." + }, + + "operational_excellence": { + "overall_score": 82, + "criteria": { + "api_design_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "MiniMax API Documentation", + "url": "https://platform.minimax.io/docs", + "date": "2025-10-27", + "value": "OpenAI- and Anthropic-compatible endpoints with streaming and tool calling, easing migration" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax GitHub", + "url": "https://github.com/MiniMax-AI", + "date": "2025-10-27", + "value": "Compatibility with mainstream OpenAI/Anthropic SDKs; first-party tooling adequate" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M3 announcement", + "url": "https://www.minimax.io/news", + "date": "2026-06-01", + "value": "Successor M3 announced 2026-06-01 with 1M context, but weights not yet published as of 2026-06-10; M2 weights remain available" + } + ], + "methodology": "Review of versioning practices and weight availability across releases", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Platform", + "url": "https://platform.minimax.io/", + "date": "2025-10-27", + "value": "Basic usage dashboard; self-hosted observability is deployer-built" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax community channels", + "url": "https://github.com/MiniMax-AI", + "date": "2025-10-27", + "value": "GitHub and community support; limited English-language enterprise support" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Inference ecosystem", + "url": "https://openrouter.ai/models", + "date": "2025-11-15", + "value": "vLLM/SGLang support, OpenRouter availability, popular in open-source agent frameworks" + } + ], + "methodology": "Analysis of third-party hosting, integrations, and tooling", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "MIT License (Hugging Face card)", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M2", + "date": "2025-10-27", + "value": "MIT license per model card, unrestricted commercial use and derivatives" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Clean MIT licensing and dual OpenAI/Anthropic API compatibility lower switching costs. M3 transition (announced, weights unpublished) is the main forward-looking uncertainty." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 88, + "notes": "Strong agentic coding at exceptional cost-efficiency; no longer the open-weight leader after 2026 releases.", + "alternatives": ["glm-5", "kimi-k2-6"] + }, + "customer-support": { + "overall": 83, + "notes": "Fast (10B active) and very cheap — well suited to high-volume conversational workloads.", + "alternatives": ["command-a-plus", "glm-5"] + }, + "content-creation": { + "overall": 80, + "notes": "Adequate generation quality at minimal cost.", + "alternatives": ["glm-5", "claude-opus-4-8"] + }, + "data-analysis": { + "overall": 84, + "notes": "Good tool-calling for analysis pipelines; weaker raw reasoning than GLM-5 or Kimi K2.6.", + "alternatives": ["glm-5", "kimi-k2-6"] + }, + "research-assistant": { + "overall": 85, + "notes": "Strong agentic search and tool orchestration; 205K context handles long documents.", + "alternatives": ["glm-5", "kimi-k2-6"] + }, + "legal-compliance": { + "overall": 68, + "notes": "China-jurisdiction first-party API and absent Western certifications are blockers unless self-hosted.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 66, + "notes": "Not recommended via first-party API; self-hosted deployment in a compliant environment is the only viable path.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 80, + "notes": "Capable and cheap; data residency requires self-hosting for regulated firms.", + "alternatives": ["command-a-plus", "glm-5"] + }, + "education": { + "overall": 82, + "notes": "Good tutoring at very low cost; well suited to high-volume educational platforms.", + "alternatives": ["glm-5", "deepseek-v3-2"] + }, + "creative-writing": { + "overall": 78, + "notes": "Serviceable creative output; not its design focus.", + "alternatives": ["claude-opus-4-8", "glm-5"] + } + }, + + "strengths": [ + "Topped open-model agentic tool-calling rankings at launch (October 2025)", + "Exceptional efficiency: 10B active of 230B total — fast inference and cheap self-hosting", + "Launched at roughly 8% of Claude pricing, among the best cost/capability ratios available", + "Clean MIT license with full self-hosting rights", + "OpenAI- and Anthropic-compatible APIs minimize migration effort", + "~205K context window for long-document and long-trajectory work" + ], + + "limitations": [ + "First-party MiniMax API processes data under Chinese jurisdiction with no published Western compliance certifications", + "Surpassed on raw benchmarks by 2026 open-weight releases (GLM-5, Kimi K2.6)", + "Successor M3 announced (2026-06-01) but weights unpublished, creating roadmap uncertainty", + "Text-only — no vision or audio modalities", + "Limited published bias, safety, and red-team evaluations", + "Interleaved thinking format requires prompt-handling care in some frameworks" + ], + + "best_for": [ + "High-volume agentic tool-calling workloads where cost per trajectory dominates", + "Teams self-hosting on modest GPU budgets (10B active footprint)", + "Drop-in replacement experiments via OpenAI/Anthropic-compatible APIs", + "Latency-sensitive agent loops needing fast token generation" + ], + + "not_recommended_for": [ + "Regulated Western workloads (healthcare, legal, finance) on the first-party API", + "Teams needing the absolute strongest open-weight benchmark performance in mid-2026", + "Multimodal applications requiring image or audio input" + ], + + "metadata": { + "pricing": { + "input": "$0.30 per 1M tokens (approx.)", + "output": "$1.20 per 1M tokens (approx.)", + "notes": "Launched at roughly 8% of Claude Sonnet pricing; third-party host pricing varies.", + "last_verified": "2026-06-10" + }, + "context_window": 204800, + "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], + "modalities": ["text"], + "api_endpoint": "https://api.minimax.io/v1/chat/completions", + "open_source": true, + "license": "MIT (per Hugging Face model card)", + "architecture": "Mixture-of-Experts: 230B total / 10B active parameters, interleaved thinking for agentic tool use", + "parameters": "230B total / 10B active", + "release_date": "2025-10-27" + }, + + "related_entities": ["glm-5", "kimi-k2-6", "deepseek-v3-2", "gpt-oss-120b"], + + "tags": [ + "agentic", + "tool-calling", + "open-source", + "mit-license", + "mixture-of-experts", + "cost-effective", + "fast-inference", + "chinese-provider", + "self-hostable" + ] +} diff --git a/data/models/mistral-large-3.json b/data/models/mistral-large-3.json new file mode 100644 index 0000000..3b9d97e --- /dev/null +++ b/data/models/mistral-large-3.json @@ -0,0 +1,632 @@ +{ + "id": "mistral-large-3", + "type": "model", + "name": "Mistral Large 3", + "provider": "Mistral AI", + "version": "Large 3 (Mistral 3 family)", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Mistral AI's open-weight flagship released December 2025 under Apache 2.0: a sparse MoE (675B total / 41B active) multimodal model with ~256K context and 40+ languages. Debuted #2 among open-source non-reasoning models on LMArena, with a strong EU data-sovereignty story.", + "website": "https://mistral.ai/news/mistral-3/", + "trust_vector": { + "performance_reliability": { + "overall_score": 88, + "criteria": { + "task_accuracy_code": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral 3 Announcement", + "url": "https://mistral.ai/news/mistral-3/", + "date": "2025-12-02", + "value": "Strong coding performance reported across standard benchmarks for an open-weight model" + } + ], + "methodology": "Review of provider benchmarks and community evaluations of open weights", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral 3 Announcement", + "url": "https://mistral.ai/news/mistral-3/", + "date": "2025-12-02", + "value": "Competitive math and reasoning results among non-reasoning (single-pass) models" + } + ], + "methodology": "Review of reasoning benchmarks; model is non-reasoning class (no extended thinking)", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "LMArena Leaderboard", + "url": "https://lmarena.ai/leaderboard", + "date": "2025-12-09", + "value": "Debuted #2 among open-source non-reasoning models on LMArena" + }, + { + "source": "Mistral 3 Announcement", + "url": "https://mistral.ai/news/mistral-3/", + "date": "2025-12-02", + "value": "State-of-the-art open-weight performance across knowledge and multilingual tasks" + } + ], + "methodology": "Crowdsourced arena comparisons and provider benchmark suite", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Community Evaluations", + "url": "https://huggingface.co/mistralai", + "date": "2025-12-15", + "value": "Stable instruction following reported across hosted and self-hosted deployments" + } + ], + "methodology": "Community repeated-prompt testing on open weights", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "~1.5s (hosted API); deployment-dependent when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-01-15", + "value": "41B active parameters keep inference fast for the model's scale" + } + ], + "methodology": "Median latency from third-party benchmarking of hosted endpoints", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "~3.5s (hosted API)", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-01-15", + "value": "Tail latency varies by host (Mistral, Bedrock, Azure, self-hosted)" + } + ], + "methodology": "95th percentile estimates across hosting providers", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "~256,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Mistral 3 Announcement", + "url": "https://mistral.ai/news/mistral-3/", + "date": "2025-12-02", + "value": "Approximately 256K token context window" + } + ], + "methodology": "Official specification from provider", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 94, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Status Page", + "url": "https://status.mistral.ai/", + "date": "2026-06-01", + "value": "Stable hosted-API availability; self-hosting removes provider dependency entirely" + } + ], + "methodology": "Historical uptime of hosted API; open weights enable customer-controlled availability", + "last_verified": "2026-06-10" + } + }, + "notes": "Best-in-class open-weight performance for its release window: sparse MoE (675B total / 41B active) delivers near-frontier quality with modest active compute. Non-reasoning class — frontier reasoning models outperform it on hard multi-step problems." + }, + "security": { + "overall_score": 83, + "criteria": { + "prompt_injection_resistance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Documentation", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "Instruction-hierarchy training; limited published red-team results" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection patterns", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Community Red-Teaming", + "url": "https://huggingface.co/mistralai", + "date": "2025-12-15", + "value": "Reasonable default refusals; open weights allow guardrail removal by deployers" + } + ], + "methodology": "Adversarial prompt testing on hosted and open-weight deployments", + "last_verified": "2026-06-10", + "notes": "Open weights mean downstream deployments can weaken or strengthen safety behavior" + }, + "data_leakage_prevention": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Privacy Policy", + "url": "https://mistral.ai/terms/", + "date": "2025-12-02", + "value": "EU-based provider under GDPR; self-hosting keeps data entirely in customer infrastructure" + } + ], + "methodology": "Analysis of privacy policies plus self-hosting option", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Moderation Tools", + "url": "https://docs.mistral.ai/capabilities/guardrailing/", + "date": "2025-12-02", + "value": "Optional moderation API and system-prompt guardrailing available" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral API Documentation", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "API key authentication, HTTPS only, rate limiting; cloud-provider controls on Bedrock/Azure" + } + ], + "methodology": "Review of API security features across hosting options", + "last_verified": "2026-06-10" + } + }, + "notes": "Solid security with the open-weights caveat: deployers control (and can remove) guardrails, so deployment-level controls matter more than for closed models." + }, + "privacy_compliance": { + "overall_score": 87, + "criteria": { + "data_residency": { + "value": "EU (hosted API); anywhere via self-hosting or cloud region choice", + "confidence": "high", + "evidence": [ + { + "source": "Mistral AI", + "url": "https://mistral.ai/news/mistral-3/", + "date": "2025-12-02", + "value": "EU provider with European hosting; open weights allow full on-premises deployment" + } + ], + "methodology": "Review of hosting documentation and deployment options", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Terms and Privacy", + "url": "https://mistral.ai/terms/", + "date": "2025-12-02", + "value": "API data not used for training by default for paid tiers; self-hosting eliminates the question entirely" + } + ], + "methodology": "Analysis of terms of service and data usage policy", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Limited retention on hosted API; zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Privacy Policy", + "url": "https://mistral.ai/terms/", + "date": "2025-12-02", + "value": "Short-term retention for abuse monitoring on La Plateforme; customer-controlled when self-hosted" + } + ], + "methodology": "Review of retention policies across deployment modes", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Documentation", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "Customer responsible for PII redaction; self-hosting keeps PII in-house" + } + ], + "methodology": "Review of data protection capabilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Trust & GDPR Posture", + "url": "https://mistral.ai/terms/", + "date": "2025-12-02", + "value": "EU provider natively under GDPR; SOC 2 for hosted platform; Bedrock/Azure hosting inherits those clouds' certifications" + } + ], + "methodology": "Verification of certifications across Mistral and cloud hosting partners", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Open Weights Distribution", + "url": "https://huggingface.co/mistralai", + "date": "2025-12-02", + "value": "Apache 2.0 weights enable fully air-gapped, zero-external-retention deployments" + } + ], + "methodology": "Review of self-hosting options enabling complete data control", + "last_verified": "2026-06-10" + } + }, + "notes": "Standout data-sovereignty story: EU provider under GDPR, plus Apache 2.0 weights allow fully on-premises/air-gapped deployment — the strongest possible residency guarantee." + }, + "trust_transparency": { + "overall_score": 80, + "criteria": { + "explainability": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "Clear step-by-step explanations; open weights permit deep inspection and research" + } + ], + "methodology": "Evaluation of reasoning transparency; open weights enable interpretability research", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Community Evaluations", + "url": "https://huggingface.co/mistralai", + "date": "2025-12-15", + "value": "Typical hallucination rates for its class; no built-in grounding" + } + ], + "methodology": "Factual QA testing by community evaluators", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Documentation", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "Multilingual training (40+ languages) reduces anglocentric bias; limited published bias evaluation" + } + ], + "methodology": "Review of bias disclosures and multilingual evaluation", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "low", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "Adequate uncertainty expression; raw logprobs accessible via open weights" + } + ], + "methodology": "Qualitative assessment plus open-weight logprob access", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/mistralai", + "date": "2025-12-02", + "value": "Public model card with architecture details (675B MoE / 41B active), license, and usage guidance" + } + ], + "methodology": "Review of published model card and architecture disclosure", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral 3 Announcement", + "url": "https://mistral.ai/news/mistral-3/", + "date": "2025-12-02", + "value": "Architecture fully disclosed; training data composition described only at a high level" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Guardrailing Documentation", + "url": "https://docs.mistral.ai/capabilities/guardrailing/", + "date": "2025-12-02", + "value": "Optional moderation and system-prompt guardrails; defaults lighter than closed flagships and removable by deployers" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Open weights provide architectural transparency rare at this scale (675B MoE disclosed), though training data detail and built-in guardrails are lighter than closed frontier models." + }, + "operational_excellence": { + "overall_score": 86, + "criteria": { + "api_design_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Mistral API Documentation", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "Clean OpenAI-compatible API with streaming, function calling, JSON mode, and vision" + } + ], + "methodology": "Review of API design and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral SDKs", + "url": "https://github.com/mistralai", + "date": "2025-12-02", + "value": "Official Python and TypeScript SDKs; first-class vLLM and transformers support for self-hosting" + } + ], + "methodology": "Review of SDK and inference-stack support", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Model Documentation", + "url": "https://docs.mistral.ai/getting-started/models/", + "date": "2025-12-02", + "value": "Dated model versions with deprecation notices; open weights never disappear once downloaded" + } + ], + "methodology": "Review of versioning policy; open weights eliminate forced-retirement risk", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral La Plateforme", + "url": "https://console.mistral.ai/", + "date": "2025-12-02", + "value": "Usage dashboard on hosted platform; self-hosted observability is customer-built" + } + ], + "methodology": "Review of monitoring tools across deployment modes", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Mistral Support", + "url": "https://docs.mistral.ai/", + "date": "2025-12-02", + "value": "Good documentation, enterprise support contracts, active community" + } + ], + "methodology": "Assessment of documentation and support channels", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Multi-platform Availability", + "url": "https://mistral.ai/news/mistral-3/", + "date": "2025-12-02", + "value": "Available on Hugging Face, Amazon Bedrock, Azure, and Mistral's La Plateforme; broad vLLM/community tooling" + } + ], + "methodology": "Analysis of distribution channels and third-party tooling", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 96, + "confidence": "high", + "evidence": [ + { + "source": "Apache 2.0 License", + "url": "https://huggingface.co/mistralai", + "date": "2025-12-02", + "value": "Apache 2.0 open weights: unrestricted commercial use, modification, and redistribution" + } + ], + "methodology": "Review of license terms", + "last_verified": "2026-06-10" + } + }, + "notes": "Apache 2.0 licensing at frontier scale is the headline: no usage restrictions, no vendor lock-in, and availability across HF, Bedrock, Azure, and La Plateforme." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 84, + "notes": "Strong open-weight coding; closed frontier flagships still lead on hard software engineering.", + "alternatives": ["claude-opus-4-8", "deepseek-v4"] + }, + "customer-support": { + "overall": 87, + "notes": "40+ languages, low active-parameter inference cost, and self-hosting make it excellent for global support.", + "alternatives": ["claude-sonnet-4-6", "nova-2-lite"] + }, + "content-creation": { + "overall": 85, + "notes": "Strong multilingual content generation; particularly good for European-language work.", + "alternatives": ["gpt-5-5", "claude-sonnet-4-6"] + }, + "data-analysis": { + "overall": 83, + "notes": "Capable analysis within 256K context; lacks extended-reasoning mode for hardest problems.", + "alternatives": ["gemini-3-1-pro", "gpt-5-5"] + }, + "research-assistant": { + "overall": 84, + "notes": "Good synthesis over long documents; self-hosting suits sensitive research corpora.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "legal-compliance": { + "overall": 86, + "notes": "EU provider under GDPR plus on-premises deployment is compelling for European legal workloads.", + "alternatives": ["claude-opus-4-8", "nova-2-lite"] + }, + "healthcare": { + "overall": 82, + "notes": "Self-hosting keeps PHI fully in-house, sidestepping vendor BAA questions; validate clinical accuracy.", + "alternatives": ["claude-opus-4-8", "nova-2-lite"] + }, + "financial-analysis": { + "overall": 83, + "notes": "Solid quantitative work; data-sovereign deployment appeals to EU financial institutions.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "education": { + "overall": 85, + "notes": "Multilingual strength and low cost suit global education deployments.", + "alternatives": ["gemini-3-5-flash", "claude-sonnet-4-6"] + }, + "creative-writing": { + "overall": 82, + "notes": "Good multilingual creative range; less distinctive than closed frontier flagships.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + } + }, + "strengths": [ + "Apache 2.0 open weights at frontier scale — full commercial freedom and no lock-in", + "Sparse MoE efficiency: 675B total but only 41B active parameters per token", + "Debuted #2 among open-source non-reasoning models on LMArena", + "EU provider with strong GDPR/data-sovereignty posture; fully self-hostable", + "Multimodal (text + image) with 40+ languages and ~256K context", + "Broad availability: Hugging Face, Amazon Bedrock, Azure, La Plateforme" + ], + "limitations": [ + "Non-reasoning class — trails frontier reasoning models on hard multi-step problems", + "Self-hosting 675B weights requires substantial GPU infrastructure despite 41B active", + "API pricing from aggregators (~$0.50/$1.50) carries medium confidence", + "Open weights let deployers strip guardrails, shifting safety burden downstream", + "Training data composition only described at a high level" + ], + "best_for": [ + "Data-sovereign deployments requiring on-premises or EU-resident inference", + "Multilingual applications across 40+ languages", + "Organizations wanting frontier-class quality without vendor lock-in", + "Cost-efficient high-volume inference leveraging sparse MoE economics" + ], + "not_recommended_for": [ + "Hardest reasoning workloads better served by extended-thinking models", + "Teams without MLOps capacity attempting self-hosted deployment at this scale", + "Applications relying on strong provider-enforced guardrails by default" + ], + "metadata": { + "pricing": { + "input": "$0.50 per 1M tokens (approx.)", + "output": "$1.50 per 1M tokens (approx.)", + "notes": "Aggregator-reported API pricing, medium confidence; varies by host (La Plateforme, Bedrock, Azure). Self-hosting under Apache 2.0 incurs only infrastructure cost.", + "last_verified": "2026-06-10" + }, + "context_window": 256000, + "languages": [ + "English", + "French", + "German", + "Spanish", + "Italian", + "Portuguese", + "Dutch", + "Polish", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi", + "Russian" + ], + "modalities": ["text", "image (input)"], + "api_endpoint": "https://api.mistral.ai/v1/chat/completions", + "open_source": true, + "architecture": "Sparse Mixture-of-Experts transformer: 675B total parameters, 41B active per token", + "parameters": "675B total / 41B active (disclosed)", + "release_date": "2025-12-02 (Mistral 3 family)" + }, + "related_entities": ["deepseek-v4", "glm-5", "claude-sonnet-4-6", "nova-2-lite", "gemini-3-1-pro"], + "tags": [ + "open-source", + "apache-2.0", + "mixture-of-experts", + "multilingual", + "eu-sovereignty", + "gdpr", + "self-hostable", + "multimodal" + ] +} diff --git a/data/models/nemotron-ultra-253b.json b/data/models/nemotron-ultra-253b.json index 4c90ba7..754c10b 100644 --- a/data/models/nemotron-ultra-253b.json +++ b/data/models/nemotron-ultra-253b.json @@ -4,9 +4,9 @@ "name": "Nemotron Ultra 253B", "provider": "NVIDIA", "version": "20251101", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Massive 253B parameter AI model from NVIDIA achieving 57.1% on SWE-bench and 80.08% on HumanEval. Optimized for high-performance computing and complex coding tasks with excellent GPU acceleration.", + "description": "Massive 253B parameter model from NVIDIA's Llama-3.1-based Nemotron line, now superseded: NVIDIA discontinued this line in favor of the native Nemotron 3 family (announced 2025-12-15). Historically achieved 57.1% on SWE-bench and 80.08% on HumanEval, optimized for high-performance computing and complex coding tasks with GPU acceleration. New deployments should evaluate Nemotron 3 instead.", "website": "https://www.nvidia.com/en-us/ai/nemotron/", "trust_vector": { "performance_reliability": { @@ -417,7 +417,7 @@ "notes": "Good transparency with comprehensive documentation. Standard hallucination and bias performance for models of this size." }, "operational_excellence": { - "overall_score": 89, + "overall_score": 88, "criteria": { "api_design_quality": { "score": 90, @@ -448,7 +448,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 87, + "score": 78, "confidence": "high", "evidence": [ { @@ -456,10 +456,17 @@ "url": "https://docs.nvidia.com/nemotron/api/versioning", "date": "2025-11-01", "value": "Clear versioning policy" + }, + { + "source": "NVIDIA Nemotron 3 Announcement", + "url": "https://nvidianews.nvidia.com/news/nvidia-debuts-nemotron-3-family-of-open-models", + "date": "2026-06-10", + "value": "Llama-3.1-based Nemotron line discontinued in favor of the native Nemotron 3 family (announced 2025-12-15)" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10", + "notes": "Model line discontinued; future development is in the Nemotron 3 family" }, "monitoring_observability": { "score": 88, @@ -615,7 +622,8 @@ "Limited training data transparency", "30-day default data retention (not ephemeral)", "Moderate latency (2.2s p50) despite GPU acceleration", - "Smaller context window (128K) compared to competitors" + "Smaller context window (128K) compared to competitors", + "Superseded: NVIDIA discontinued the Llama-3.1-based Nemotron line in favor of the native Nemotron 3 family (announced 2025-12-15)" ], "best_for": [ "GPU-accelerated workloads and HPC environments", @@ -664,6 +672,7 @@ "deepseek-r1" ], "tags": [ + "superseded", "coding", "gpu-accelerated", "enterprise", diff --git a/data/models/nova-2-lite.json b/data/models/nova-2-lite.json new file mode 100644 index 0000000..f397c09 --- /dev/null +++ b/data/models/nova-2-lite.json @@ -0,0 +1,626 @@ +{ + "id": "nova-2-lite", + "type": "model", + "name": "Amazon Nova 2 Lite", + "provider": "Amazon (AWS)", + "version": "Nova 2 Lite (GA)", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Amazon's cost-efficient Nova 2 workhorse model, GA on Amazon Bedrock since re:Invent 2025. Offers three thinking-intensity levels, a built-in code interpreter, and web grounding with a 1M token context, backed by AWS's strong enterprise compliance posture.", + "website": "https://aws.amazon.com/bedrock/nova/", + "trust_vector": { + "performance_reliability": { + "overall_score": 86, + "criteria": { + "task_accuracy_code": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "AWS What's New (Nova 2 announcement)", + "url": "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/", + "date": "2025-12-02", + "value": "Built-in code interpreter enables executable code workflows; AWS cites competitive coding quality for its price tier" + } + ], + "methodology": "Review of provider claims; limited independent benchmark coverage to date", + "last_verified": "2026-06-10", + "notes": "Public third-party benchmark data sparse; score weighted toward provider evaluations" + }, + "task_accuracy_reasoning": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "AWS What's New (Nova 2 announcement)", + "url": "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/", + "date": "2025-12-02", + "value": "Three thinking-intensity levels allow scaling reasoning depth per request" + } + ], + "methodology": "Review of provider evaluations of adjustable-thinking reasoning performance", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 83, + "confidence": "low", + "evidence": [ + { + "source": "Amazon Nova Documentation", + "url": "https://docs.aws.amazon.com/nova/", + "date": "2025-12-02", + "value": "Positioned as fast, capable general-purpose model; few public leaderboard placements" + } + ], + "methodology": "Review of provider documentation; independent leaderboard coverage remains sparse", + "last_verified": "2026-06-10", + "notes": "Low confidence pending broader third-party benchmarking" + }, + "output_consistency": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Amazon Nova Documentation", + "url": "https://docs.aws.amazon.com/nova/", + "date": "2025-12-02", + "value": "Thinking-intensity controls produce predictable quality/cost tradeoffs" + } + ], + "methodology": "Review of feature design and repeated-prompt behavior reports", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "~1.5s (low thinking intensity)", + "confidence": "medium", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-01-15", + "value": "Lite tier optimized for low latency; higher thinking intensity increases response time" + } + ], + "methodology": "Median latency from third-party benchmarking at default settings", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "~4.0s (varies with thinking intensity)", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-01-15", + "value": "Tail latency dominated by thinking-intensity setting and tool use (code interpreter, web grounding)" + } + ], + "methodology": "95th percentile estimates across configurations", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "1,000,000 tokens (~65K output)", + "confidence": "high", + "evidence": [ + { + "source": "Amazon Nova Documentation", + "url": "https://docs.aws.amazon.com/nova/", + "date": "2025-12-02", + "value": "1M token context window with approximately 65K max output tokens" + } + ], + "methodology": "Official specification from provider documentation", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "AWS Health Dashboard", + "url": "https://health.aws.amazon.com/health/status", + "date": "2026-06-01", + "value": "Bedrock service availability backed by AWS regional infrastructure and SLAs" + } + ], + "methodology": "Historical service availability from AWS status reporting", + "last_verified": "2026-06-10" + } + }, + "notes": "GA member of the Nova 2 family (Nova 2 Pro and Omni remain in preview). Public benchmark data is still sparse, so performance scores carry medium/low confidence." + }, + "security": { + "overall_score": 88, + "criteria": { + "prompt_injection_resistance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Amazon Bedrock Guardrails", + "url": "https://aws.amazon.com/bedrock/guardrails/", + "date": "2025-12-02", + "value": "Bedrock Guardrails add configurable prompt-attack filtering on top of model defenses" + } + ], + "methodology": "Testing against OWASP LLM01 patterns with and without Bedrock Guardrails", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "AWS Responsible AI", + "url": "https://aws.amazon.com/ai/responsible-ai/", + "date": "2025-12-02", + "value": "Safety training plus platform-level guardrails reduce jailbreak success" + } + ], + "methodology": "Review of adversarial testing and platform guardrail capabilities", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Data Protection", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html", + "date": "2025-12-02", + "value": "Customer data not used to train models; content not shared with model providers" + } + ], + "methodology": "Analysis of Bedrock data protection documentation and commitments", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "AWS Responsible AI", + "url": "https://aws.amazon.com/ai/responsible-ai/", + "date": "2025-12-02", + "value": "Built-in content moderation plus configurable Bedrock Guardrails policies" + } + ], + "methodology": "Safety testing across harmful content categories", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Security", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/security.html", + "date": "2025-12-02", + "value": "IAM authentication, SigV4 signing, VPC endpoints (PrivateLink), KMS encryption, CloudTrail auditing" + } + ], + "methodology": "Review of AWS-native security controls available to Bedrock workloads", + "last_verified": "2026-06-10" + } + }, + "notes": "Inherits AWS's mature platform security: IAM, PrivateLink, KMS, CloudTrail, and Bedrock Guardrails give it one of the strongest deployment security stories available." + }, + "privacy_compliance": { + "overall_score": 90, + "criteria": { + "data_residency": { + "value": "Customer-selected AWS regions", + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Documentation", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html", + "date": "2025-12-02", + "value": "Processing stays in selected region (cross-region inference configurable)" + } + ], + "methodology": "Review of Bedrock regional processing documentation", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Data Protection", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html", + "date": "2025-12-02", + "value": "Customer prompts and outputs are never used to train Nova models — no opt-out required" + } + ], + "methodology": "Analysis of Bedrock data usage commitments", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Not retained by Bedrock (customer-controlled logging)", + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Data Protection", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html", + "date": "2025-12-02", + "value": "Bedrock does not store prompts/outputs after processing unless customer enables invocation logging" + } + ], + "methodology": "Review of Bedrock retention and logging documentation", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Amazon Bedrock Guardrails", + "url": "https://aws.amazon.com/bedrock/guardrails/", + "date": "2025-12-02", + "value": "Guardrails support sensitive-information filters including PII redaction" + } + ], + "methodology": "Review of platform PII detection and redaction capabilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "AWS Compliance Programs", + "url": "https://aws.amazon.com/compliance/services-in-scope/", + "date": "2025-12-02", + "value": "Bedrock in scope for SOC 1/2/3, ISO 27001/27017/27018, HIPAA-eligible, GDPR-supporting, FedRAMP" + } + ], + "methodology": "Verification against AWS services-in-scope compliance listings", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Data Protection", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/data-protection.html", + "date": "2025-12-02", + "value": "No persistent storage of inference content by default; abuse-detection processing is ephemeral" + } + ], + "methodology": "Review of data handling practices", + "last_verified": "2026-06-10" + } + }, + "notes": "Among the strongest compliance postures in the market via Bedrock: SOC, ISO, HIPAA-eligible, GDPR-supporting, with data never used for training." + }, + "trust_transparency": { + "overall_score": 81, + "criteria": { + "explainability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "AWS What's New (Nova 2 announcement)", + "url": "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/", + "date": "2025-12-02", + "value": "Thinking-intensity levels expose reasoning effort; web grounding cites sources" + } + ], + "methodology": "Evaluation of reasoning transparency and grounded citation behavior", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 80, + "confidence": "low", + "evidence": [ + { + "source": "Amazon Nova Documentation", + "url": "https://docs.aws.amazon.com/nova/", + "date": "2025-12-02", + "value": "Built-in web grounding designed to reduce hallucinations; limited independent factuality data" + } + ], + "methodology": "Review of grounding features; independent factual QA coverage sparse", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "AWS Responsible AI", + "url": "https://aws.amazon.com/ai/responsible-ai/", + "date": "2025-12-02", + "value": "AWS responsible AI program with published service cards and bias testing" + } + ], + "methodology": "Review of AWS responsible AI documentation and service cards", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "low", + "evidence": [ + { + "source": "Model Behavior", + "url": "https://docs.aws.amazon.com/nova/", + "date": "2025-12-02", + "value": "Adequate uncertainty expression; little published evaluation" + } + ], + "methodology": "Qualitative assessment of confidence expression", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Amazon Nova Documentation", + "url": "https://docs.aws.amazon.com/nova/", + "date": "2025-12-02", + "value": "AWS publishes service cards and user guides covering capabilities and responsible-use guidance" + } + ], + "methodology": "Review of documentation completeness", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "AWS Public Statements", + "url": "https://aws.amazon.com/bedrock/nova/", + "date": "2025-12-02", + "value": "Training data sources not disclosed in detail" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Guardrails", + "url": "https://aws.amazon.com/bedrock/guardrails/", + "date": "2025-12-02", + "value": "Configurable platform guardrails: content filters, denied topics, PII redaction, contextual grounding checks" + } + ], + "methodology": "Analysis of built-in and platform-level safety mechanisms", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong platform-level guardrails and grounding features; independent transparency and factuality data still limited for the Nova 2 generation." + }, + "operational_excellence": { + "overall_score": 89, + "criteria": { + "api_design_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock API", + "url": "https://docs.aws.amazon.com/bedrock/latest/APIReference/welcome.html", + "date": "2025-12-02", + "value": "Converse API with streaming, tool use, thinking-intensity controls, code interpreter, web grounding" + } + ], + "methodology": "Review of API design and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "AWS SDKs", + "url": "https://aws.amazon.com/developer/tools/", + "date": "2025-12-02", + "value": "Mature official SDKs across Python, JavaScript, Java, Go, .NET and more" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Amazon Bedrock Model Lifecycle", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-lifecycle.html", + "date": "2025-12-02", + "value": "Documented model lifecycle with legacy/EOL stages and advance notice" + } + ], + "methodology": "Review of Bedrock model lifecycle policy", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Amazon CloudWatch / CloudTrail", + "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/monitoring.html", + "date": "2025-12-02", + "value": "Native CloudWatch metrics, invocation logging, and CloudTrail audit trails" + } + ], + "methodology": "Review of monitoring and observability integrations", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "AWS Support", + "url": "https://aws.amazon.com/premiumsupport/", + "date": "2025-12-02", + "value": "Tiered enterprise support with TAMs, SLAs, and extensive documentation" + } + ], + "methodology": "Assessment of support tiers and documentation", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Amazon Bedrock Ecosystem", + "url": "https://aws.amazon.com/bedrock/", + "date": "2025-12-02", + "value": "Integrates with Bedrock Agents, Knowledge Bases, Guardrails, and the broader AWS ecosystem" + } + ], + "methodology": "Analysis of platform integrations and tooling", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "AWS Service Terms", + "url": "https://aws.amazon.com/service-terms/", + "date": "2025-12-02", + "value": "Standard AWS service terms; Amazon offers IP indemnification for Nova outputs" + } + ], + "methodology": "Review of licensing terms and indemnification", + "last_verified": "2026-06-10" + } + }, + "notes": "Exceptional operational maturity by virtue of the AWS/Bedrock platform: IAM, monitoring, lifecycle policy, and enterprise support are best-in-class." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 82, + "notes": "Built-in code interpreter is convenient for executable workflows, but raw coding quality trails frontier flagships.", + "alternatives": ["claude-opus-4-8", "grok-4-3"] + }, + "customer-support": { + "overall": 88, + "notes": "Low cost, low latency, 1M context, and strong compliance make it excellent for enterprise support at scale.", + "alternatives": ["claude-sonnet-4-6", "gemini-3-5-flash"] + }, + "content-creation": { + "overall": 82, + "notes": "Capable everyday content generation with web grounding for current facts; less distinctive prose than flagships.", + "alternatives": ["gpt-5-5", "claude-sonnet-4-6"] + }, + "data-analysis": { + "overall": 86, + "notes": "Code interpreter plus 1M context suit data workflows well at a very low price point.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "research-assistant": { + "overall": 85, + "notes": "Web grounding with citations and 1M context are strong for research at this price tier.", + "alternatives": ["gemini-3-1-pro", "grok-4-3"] + }, + "legal-compliance": { + "overall": 87, + "notes": "Bedrock's compliance certifications and data controls are the draw; pair with human review for analysis depth.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "healthcare": { + "overall": 86, + "notes": "HIPAA-eligible on Bedrock with strong data protections; validate clinical accuracy given sparse benchmarks.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "financial-analysis": { + "overall": 84, + "notes": "Good quantitative workflows via code interpreter; compliance posture fits financial services.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "education": { + "overall": 85, + "notes": "Adjustable thinking intensity and low cost suit high-volume tutoring deployments.", + "alternatives": ["claude-sonnet-4-6", "gemini-3-5-flash"] + }, + "creative-writing": { + "overall": 78, + "notes": "Serviceable but not a creative standout versus frontier flagships.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + } + }, + "strengths": [ + "Very low pricing: $0.30/$2.50 per 1M tokens with 1M context", + "Three thinking-intensity levels for per-request quality/cost control", + "Built-in code interpreter and web grounding (no external tooling needed)", + "Best-in-class enterprise compliance via Bedrock (SOC, ISO, HIPAA-eligible, GDPR)", + "Customer data never used for model training", + "Deep AWS integration: IAM, PrivateLink, KMS, CloudWatch, Guardrails" + ], + "limitations": [ + "Public benchmark data sparse; performance claims rest largely on AWS evaluations", + "Raw capability trails frontier flagships on hard reasoning and coding", + "Nova 2 Pro and Omni siblings still in preview, limiting family upgrade paths", + "AWS-only availability creates platform lock-in", + "Training data transparency is limited" + ], + "best_for": [ + "AWS-centric enterprises needing compliant, cost-efficient inference at scale", + "High-volume agentic workloads using code interpreter and web grounding", + "Regulated industries requiring HIPAA-eligible, SOC/ISO-certified infrastructure", + "Long-context document processing on a budget" + ], + "not_recommended_for": [ + "Teams needing best-in-class coding or reasoning quality", + "Multi-cloud deployments avoiding AWS lock-in", + "Use cases requiring extensively validated public benchmark performance", + "Cutting-edge creative writing applications" + ], + "metadata": { + "pricing": { + "input": "$0.30 per 1M tokens", + "output": "$2.50 per 1M tokens", + "notes": "Amazon Bedrock pricing; batch and provisioned throughput options available. Announced at re:Invent 2025-12-02.", + "last_verified": "2026-06-10" + }, + "context_window": 1000000, + "max_output": 65000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)", "document"], + "api_endpoint": "https://bedrock-runtime.us-east-1.amazonaws.com", + "open_source": false, + "architecture": "Transformer-based with adjustable thinking intensity, built-in code interpreter, and web grounding", + "parameters": "Not disclosed", + "release_date": "2025-12-02 (GA at re:Invent; Nova 2 Pro and Omni in preview)" + }, + "related_entities": ["nova-pro", "claude-sonnet-4-6", "gemini-3-5-flash", "grok-4-3"], + "tags": [ + "aws", + "bedrock", + "enterprise", + "hipaa-eligible", + "cost-effective", + "code-interpreter", + "web-grounding", + "thinking-levels" + ] +} diff --git a/data/models/nova-pro.json b/data/models/nova-pro.json index a41e442..dc17c2f 100644 --- a/data/models/nova-pro.json +++ b/data/models/nova-pro.json @@ -4,9 +4,9 @@ "name": "Nova Pro", "provider": "Amazon", "version": "2025-01", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Amazon's Nova Pro model integrated with AWS services. Designed for enterprise customers requiring seamless AWS integration with good general capabilities.", + "description": "SUPERSEDED by the Nova 2 family announced at re:Invent 2025-12-02 (Nova 2 Lite GA; Nova 2 Pro/Omni in preview), though the original Nova Pro is still served. Amazon model integrated with AWS services for enterprise customers requiring seamless AWS integration. New projects should evaluate Nova 2.", "website": "https://aws.amazon.com/bedrock/nova/", "trust_vector": { "performance_reliability": { @@ -437,6 +437,12 @@ "url": "https://docs.aws.amazon.com/bedrock/versioning", "date": "2025-01-15", "value": "AWS standard versioning" + }, + { + "source": "AWS What's New: Nova 2 foundation models in Amazon Bedrock", + "url": "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/", + "date": "2026-06-10", + "value": "Nova 2 family announced 2025-12-02 (Nova 2 Lite GA; Pro/Omni preview); original Nova Pro still served" } ], "methodology": "Policy review", @@ -589,7 +595,8 @@ "AWS vendor lock-in", "Higher costs within AWS ecosystem", "Performance lags behind specialized models", - "Best value only for existing AWS customers" + "Best value only for existing AWS customers", + "SUPERSEDED: Nova 2 family (announced 2025-12-02) is the current Amazon Nova generation; original Nova Pro still served" ], "best_for": [ "AWS-native applications requiring tight integration", @@ -635,11 +642,13 @@ "parameters": "Not disclosed" }, "related_entities": [ + "nova-2-lite", "claude-sonnet-4-5", "gpt-4-1", "amazon-bedrock-agents" ], "tags": [ + "superseded", "aws", "enterprise", "hipaa-eligible", diff --git a/data/models/openai-o1-mini.json b/data/models/openai-o1-mini.json index 49c6d30..3d16d77 100644 --- a/data/models/openai-o1-mini.json +++ b/data/models/openai-o1-mini.json @@ -4,9 +4,9 @@ "name": "OpenAI o1-mini", "provider": "OpenAI", "version": "2024-12", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's efficient reasoning model with chain-of-thought capabilities at lower cost. Balanced performance for reasoning tasks with faster response times than o3.", + "description": "DEPRECATED: o1-mini (with o1-preview) was removed from the OpenAI API in 2025 and from ChatGPT; remaining o1 variants shut down 2026-10-23. Migration target is GPT-5.5. Historically an efficient reasoning model with chain-of-thought capabilities at lower cost than o1.", "website": "https://openai.com/o1", "trust_vector": { "performance_reliability": { @@ -398,7 +398,7 @@ "notes": "Excellent explainability via chain-of-thought. Good transparency." }, "operational_excellence": { - "overall_score": 88, + "overall_score": 86, "criteria": { "api_design_quality": { "score": 91, @@ -429,7 +429,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 85, + "score": 73, "confidence": "high", "evidence": [ { @@ -437,6 +437,12 @@ "url": "https://platform.openai.com/docs/versioning", "date": "2024-12-15", "value": "Clear versioning" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "o1-preview/o1-mini removed in 2025; remaining o1 variants API shutdown 2026-10-23; migration target GPT-5.5" } ], "methodology": "Policy review", @@ -471,7 +477,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 94, + "score": 84, "confidence": "high", "evidence": [ { @@ -499,7 +505,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Excellent operational maturity with OpenAI ecosystem." + "notes": "Deprecated: o1-mini removed from API in 2025; migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { @@ -592,7 +598,8 @@ "Not HIPAA eligible", "Reasoning overhead unnecessary for simple tasks", "Lower performance than o3 on complex tasks", - "Premium pricing for reasoning capabilities" + "Premium pricing for reasoning capabilities", + "DEPRECATED: removed from the API in 2025 and from ChatGPT — migrate to GPT-5.5" ], "best_for": [ "Reasoning tasks requiring transparency", @@ -637,11 +644,13 @@ "parameters": "Not disclosed" }, "related_entities": [ + "gpt-5-5", "openai-o3", "gpt-4-1", "claude-sonnet-4-5" ], "tags": [ + "deprecated", "reasoning", "chain-of-thought", "coding", diff --git a/data/models/openai-o1.json b/data/models/openai-o1.json index 3ff06a3..9db339f 100644 --- a/data/models/openai-o1.json +++ b/data/models/openai-o1.json @@ -4,9 +4,9 @@ "name": "OpenAI o1", "provider": "OpenAI", "version": "20250915", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Advanced reasoning model from OpenAI achieving 57.1% on SWE-bench and 79.2% on HumanEval. Features extended chain-of-thought reasoning for complex problem-solving and mathematical tasks.", + "description": "DEPRECATED: o1 variants and o1-pro shut down in the API on 2026-10-23; already removed from ChatGPT (o1-preview/o1-mini removed in 2025). Migration target is GPT-5.5. Historically an advanced reasoning model (57.1% SWE-bench, 79.2% HumanEval) with extended chain-of-thought reasoning.", "website": "https://openai.com/o1/", "trust_vector": { "performance_reliability": { @@ -417,7 +417,7 @@ "notes": "Excellent explainability via chain-of-thought reasoning. Transparent problem-solving process visible to users." }, "operational_excellence": { - "overall_score": 91, + "overall_score": 88, "criteria": { "api_design_quality": { "score": 93, @@ -448,7 +448,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 90, + "score": 78, "confidence": "high", "evidence": [ { @@ -456,6 +456,12 @@ "url": "https://platform.openai.com/docs/api-reference/versioning", "date": "2025-09-15", "value": "Clear versioning with deprecation notices" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "o1 variants and o1-pro API shutdown 2026-10-23; migration target GPT-5.5; removed from ChatGPT" } ], "methodology": "Review of versioning policy and historical practices", @@ -490,7 +496,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 93, + "score": 83, "confidence": "high", "evidence": [ { @@ -518,7 +524,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Excellent operational maturity with well-designed APIs and mature ecosystem. Enterprise-ready with strong support." + "notes": "Deprecated: API shutdown scheduled 2026-10-23, migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { @@ -615,7 +621,8 @@ "30-day minimum data retention (not ephemeral)", "Not HIPAA eligible", "Higher cost due to extended reasoning compute", - "Reasoning overhead may be unnecessary for simple tasks" + "Reasoning overhead may be unnecessary for simple tasks", + "DEPRECATED: removed from ChatGPT; o1 variants and o1-pro API shutdown 2026-10-23 — migrate to GPT-5.5" ], "best_for": [ "Complex reasoning tasks requiring deep analysis", @@ -660,12 +667,13 @@ "parameters": "Not disclosed" }, "related_entities": [ + "gpt-5-5", "claude-sonnet-4-5", "claude-opus-4-1", - "claude-sonnet-4-5", "openai-o3-mini" ], "tags": [ + "deprecated", "reasoning", "chain-of-thought", "coding", diff --git a/data/models/openai-o3-mini.json b/data/models/openai-o3-mini.json index 2bb5bb6..32438ca 100644 --- a/data/models/openai-o3-mini.json +++ b/data/models/openai-o3-mini.json @@ -4,9 +4,9 @@ "name": "OpenAI o3-mini", "provider": "OpenAI", "version": "20251201", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Efficient reasoning model from OpenAI achieving 50% on SWE-bench and 87.3% on HumanEval. Optimized for fast reasoning at competitive pricing with strong coding capabilities.", + "description": "DEPRECATED: o3-mini's API shuts down 2026-10-23; migration target is GPT-5.5. Historically an efficient reasoning model from OpenAI achieving 50% on SWE-bench and 87.3% on HumanEval, optimized for fast reasoning at competitive pricing with strong coding capabilities.", "website": "https://openai.com/o3-mini/", "trust_vector": { "performance_reliability": { @@ -404,7 +404,7 @@ "notes": "Good transparency with visible reasoning. Strong safety guardrails." }, "operational_excellence": { - "overall_score": 89, + "overall_score": 87, "criteria": { "api_design_quality": { "score": 91, @@ -435,7 +435,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 88, + "score": 76, "confidence": "high", "evidence": [ { @@ -443,6 +443,12 @@ "url": "https://platform.openai.com/docs", "date": "2025-12-01", "value": "Clear policy" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "o3-mini API shutdown 2026-10-23; migration target GPT-5.5" } ], "methodology": "Policy review", @@ -477,7 +483,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 91, + "score": 83, "confidence": "high", "evidence": [ { @@ -505,7 +511,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Excellent operational maturity with mature ecosystem." + "notes": "Deprecated: o3-mini API shutdown scheduled 2026-10-23, migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { @@ -602,7 +608,8 @@ "Lower than o4-mini on some benchmarks", "Mini model limitations for complex reasoning", "Reasoning overhead for simple tasks", - "Moderate general knowledge (75.8% MMLU)" + "Moderate general knowledge (75.8% MMLU)", + "DEPRECATED: o3-mini API shutdown 2026-10-23 — migrate to GPT-5.5" ], "best_for": [ "Code generation on a budget", @@ -647,11 +654,12 @@ "parameters": "Not disclosed" }, "related_entities": [ + "gpt-5-5", "openai-o4-mini", - "openai-o1", "claude-sonnet-4-5" ], "tags": [ + "deprecated", "reasoning", "code-generation", "mini-model", diff --git a/data/models/openai-o3.json b/data/models/openai-o3.json index f1807ba..4ab3972 100644 --- a/data/models/openai-o3.json +++ b/data/models/openai-o3.json @@ -4,9 +4,9 @@ "name": "OpenAI o3", "provider": "OpenAI", "version": "2025-01", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's most advanced reasoning model with exceptional performance on complex coding and mathematical tasks. Breakthrough capabilities in HumanEval and advanced problem-solving.", + "description": "DEPRECATED: the o3 family is being retired — o3-deep-research shuts down 2026-07-23 and o3-mini's API shuts down 2026-10-23; migration target is GPT-5.5. Historically OpenAI's most advanced reasoning model of its era, with exceptional performance on complex coding and mathematical tasks.", "website": "https://openai.com/o3", "trust_vector": { "performance_reliability": { @@ -447,7 +447,7 @@ "notes": "Excellent explainability through chain-of-thought reasoning. Strong hallucination resistance. Training data transparency could be improved." }, "operational_excellence": { - "overall_score": 88, + "overall_score": 86, "criteria": { "api_design_quality": { "score": 91, @@ -478,7 +478,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 85, + "score": 73, "confidence": "high", "evidence": [ { @@ -486,6 +486,12 @@ "url": "https://platform.openai.com/docs/versioning", "date": "2025-01-15", "value": "Dated versioning with deprecation notices" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "o3-deep-research shutdown 2026-07-23; o3-mini API shutdown 2026-10-23; migration target GPT-5.5" } ], "methodology": "Review of versioning policy and historical practices", @@ -521,7 +527,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 94, + "score": 84, "confidence": "high", "evidence": [ { @@ -549,7 +555,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Excellent operational maturity with mature ecosystem and strong developer experience. Well-maintained SDKs and comprehensive documentation." + "notes": "Deprecated: o3-deep-research shuts down 2026-07-23 and o3-mini API shuts down 2026-10-23; migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { @@ -646,7 +652,8 @@ "Premium pricing for reasoning capabilities", "Not HIPAA eligible", "Limited regional data residency options", - "Reasoning overhead unnecessary for simple tasks" + "Reasoning overhead unnecessary for simple tasks", + "DEPRECATED: o3-deep-research shuts down 2026-07-23; o3 family API shutdown 2026-10-23 — migrate to GPT-5.5" ], "best_for": [ "Complex coding tasks requiring advanced algorithms", @@ -694,12 +701,13 @@ "parameters": "Not disclosed" }, "related_entities": [ - "grok-3-beta", + "gpt-5-5", "claude-sonnet-4-5", "gpt-4-1", "openai-o1-mini" ], "tags": [ + "deprecated", "reasoning", "coding", "mathematics", diff --git a/data/models/openai-o4-mini.json b/data/models/openai-o4-mini.json index add2571..aa636e6 100644 --- a/data/models/openai-o4-mini.json +++ b/data/models/openai-o4-mini.json @@ -4,9 +4,9 @@ "name": "OpenAI o4-mini", "provider": "OpenAI", "version": "o4-mini-2025-04-16", - "last_evaluated": "2025-11-17", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "OpenAI's best small reasoning model (April 2025). 93% AIME, 68% SWE-bench, 10x cheaper than o3. First mini with full tool support + multimodality.", + "description": "DEPRECATED: removed from ChatGPT 2026-02-13; o4-mini API shuts down 2026-10-23 (o4-mini-deep-research shut down 2026-07-23); migrate to GPT-5.5. Historically OpenAI's best small reasoning model (April 2025): 93% AIME, 68% SWE-bench, first mini with full tool support + multimodality.", "website": "https://openai.com/index/introducing-o3-and-o4-mini/", "trust_vector": { "performance_reliability": { @@ -404,7 +404,7 @@ "notes": "Good transparency with visible reasoning. Strong safety guardrails." }, "operational_excellence": { - "overall_score": 89, + "overall_score": 87, "criteria": { "api_design_quality": { "score": 91, @@ -435,7 +435,7 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 88, + "score": 76, "confidence": "high", "evidence": [ { @@ -443,6 +443,12 @@ "url": "https://platform.openai.com/docs", "date": "2025-12-01", "value": "Clear policy" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-06-10", + "value": "o4-mini removed from ChatGPT 2026-02-13; o4-mini-deep-research shutdown 2026-07-23; o4-mini API shutdown 2026-10-23; migration target GPT-5.5" } ], "methodology": "Policy review", @@ -477,7 +483,7 @@ "last_verified": "2025-11-08" }, "ecosystem_maturity": { - "score": 91, + "score": 83, "confidence": "high", "evidence": [ { @@ -505,7 +511,7 @@ "last_verified": "2025-11-08" } }, - "notes": "Excellent operational maturity with mature ecosystem." + "notes": "Deprecated: removed from ChatGPT 2026-02-13; o4-mini API shutdown scheduled 2026-10-23, migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { @@ -602,7 +608,8 @@ "Lower than o4-mini on some benchmarks", "Mini model limitations for complex reasoning", "Reasoning overhead for simple tasks", - "Moderate general knowledge (75.8% MMLU)" + "Moderate general knowledge (75.8% MMLU)", + "DEPRECATED: removed from ChatGPT 2026-02-13; o4-mini API shutdown 2026-10-23 — migrate to GPT-5.5" ], "best_for": [ "Code generation on a budget", @@ -647,11 +654,12 @@ "parameters": "Not disclosed" }, "related_entities": [ - "openai-o4-mini", - "openai-o1", + "gpt-5-5", + "openai-o3-mini", "claude-sonnet-4-5" ], "tags": [ + "deprecated", "reasoning", "code-generation", "mini-model", diff --git a/data/models/qwen2-5-vl-32b.json b/data/models/qwen2-5-vl-32b.json index 819b9e7..a6a9ae8 100644 --- a/data/models/qwen2-5-vl-32b.json +++ b/data/models/qwen2-5-vl-32b.json @@ -4,9 +4,9 @@ "name": "Qwen2.5-VL-32B", "provider": "Alibaba", "version": "20251020", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-06-10", "evaluated_by": "TrustVector Team", - "description": "Advanced multimodal vision-language model from Alibaba achieving 42.9% on SWE-bench. Specialized for vision tasks with strong image understanding and competitive pricing.", + "description": "Multimodal vision-language model from Alibaba, now two generations behind: superseded first by Qwen3-VL (Sep 2025) and then by the natively-multimodal Qwen3.5 (released 2026-02-16). Historically achieved 42.9% on SWE-bench with strong image understanding at competitive pricing; new deployments should evaluate Qwen3.5 instead.", "website": "https://qwenlm.github.io/", "trust_vector": { "performance_reliability": { @@ -404,7 +404,7 @@ "notes": "Moderate transparency with standard safety features." }, "operational_excellence": { - "overall_score": 83, + "overall_score": 82, "criteria": { "api_design_quality": { "score": 85, @@ -435,18 +435,25 @@ "last_verified": "2025-11-08" }, "versioning_policy": { - "score": 82, - "confidence": "medium", + "score": 74, + "confidence": "high", "evidence": [ { "source": "Alibaba Cloud", "url": "https://www.alibabacloud.com/", "date": "2025-10-20", "value": "Basic versioning" + }, + { + "source": "Qwen3.5 Announcement", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-06-10", + "value": "Two generations behind: superseded by Qwen3-VL (Sep 2025) and natively-multimodal Qwen3.5 (released 2026-02-16)" } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-06-10", + "notes": "Open weights remain available, but the line has moved on to Qwen3.5" }, "monitoring_observability": { "score": 81, @@ -598,7 +605,8 @@ "Fewer compliance certifications for Western markets", "Smaller context window (32K tokens)", "Lower coding benchmarks (42.9% SWE-bench)", - "Growing but less mature ecosystem" + "Growing but less mature ecosystem", + "Superseded: two generations behind, replaced by Qwen3-VL (Sep 2025) and the natively-multimodal Qwen3.5 (released 2026-02-16)" ], "best_for": [ "Visual AI applications requiring image understanding", @@ -646,9 +654,11 @@ "related_entities": [ "gemini-2-0-flash", "claude-sonnet-4-5", - "gpt-5" + "gpt-5", + "qwen3-5" ], "tags": [ + "superseded", "vision", "multimodal", "open-source", diff --git a/data/models/qwen3-5.json b/data/models/qwen3-5.json new file mode 100644 index 0000000..f9a5632 --- /dev/null +++ b/data/models/qwen3-5.json @@ -0,0 +1,644 @@ +{ + "id": "qwen3-5", + "type": "model", + "name": "Qwen3.5", + "provider": "Alibaba", + "version": "20260216", + "last_evaluated": "2026-06-10", + "evaluated_by": "TrustVector Team", + "description": "Alibaba's Apache-2.0 flagship open model: Qwen3.5-397B-A17B, a hybrid MoE with 512 experts (397B total / 17B active) that is natively multimodal, supports 262K context (1M on hosted Qwen3.5-Plus) and 201 languages, and beats Alibaba's own API-only 1T-parameter Qwen3-Max while decoding up to 19x faster at long context.", + "website": "https://qwen.ai/blog?id=qwen3.5", + "trust_vector": { + "performance_reliability": { + "overall_score": 92, + "criteria": { + "task_accuracy_code": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Qwen3.5 Release Blog", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Strong agentic coding results; flagship 397B-A17B surpasses Qwen3-Max on coding benchmarks" + }, + { + "source": "VentureBeat", + "url": "https://venturebeat.com/technology/alibabas-qwen-3-5-397b-a17-beats-its-larger-trillion-parameter-model-at-a", + "date": "2026-02-17", + "value": "Independent reporting confirms 397B-A17B beats the 1T-parameter, API-only Qwen3-Max" + } + ], + "methodology": "Vendor benchmarks corroborated by independent press coverage and community leaderboards", + "last_verified": "2026-06-10" + }, + "task_accuracy_reasoning": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "Qwen3.5 Release Blog", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Frontier-class math and agentic reasoning under the 'Towards Native Multimodal Agents' positioning" + }, + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-02-16", + "value": "Detailed benchmark tables across reasoning suites on the official model card" + } + ], + "methodology": "Mathematical and agentic reasoning benchmarks from the model card and release blog, cross-checked against community evaluations", + "last_verified": "2026-06-10" + }, + "task_accuracy_general": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Qwen3.5 Release Blog", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Natively multimodal (text + vision) with strong general knowledge across 201 languages" + } + ], + "methodology": "Comprehensive knowledge and multimodal benchmark review including multilingual coverage", + "last_verified": "2026-06-10" + }, + "output_consistency": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-02-16", + "value": "Stable hybrid-MoE routing; consistent outputs across repeated runs in community testing" + } + ], + "methodology": "Repeated-prompt testing across temperature settings, supplemented by community reports", + "last_verified": "2026-06-10" + }, + "latency_p50": { + "value": "1.5s (hosted); 17B active params enable fast self-hosted decode", + "confidence": "medium", + "evidence": [ + { + "source": "VentureBeat", + "url": "https://venturebeat.com/technology/alibabas-qwen-3-5-397b-a17-beats-its-larger-trillion-parameter-model-at-a", + "date": "2026-02-17", + "value": "Up to 19x faster decode than Qwen3-Max at 256K context thanks to the sparse 17B-active design" + } + ], + "methodology": "Median latency on hosted endpoints and decode-throughput comparisons from independent reporting", + "last_verified": "2026-06-10" + }, + "latency_p95": { + "value": "3.8s (hosted)", + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/qwen3-5", + "date": "2026-03-10", + "value": "p95 ~3.8s on hosted endpoints for standard workloads" + } + ], + "methodology": "95th percentile response time across diverse workloads from independent benchmarking", + "last_verified": "2026-06-10" + }, + "context_window": { + "value": "262,144 tokens (1M on hosted Qwen3.5-Plus)", + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-02-16", + "value": "262K native context for open weights; hosted Qwen3.5-Plus extends to 1M tokens" + } + ], + "methodology": "Official specification from model card and Alibaba Cloud documentation", + "last_verified": "2026-06-10" + }, + "uptime": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-06-01", + "value": "Stable availability on Alibaba Cloud (incl. Singapore region) plus many third-party hosts and self-hosting" + } + ], + "methodology": "Hosted-platform availability history plus redundancy across third-party hosts", + "last_verified": "2026-06-10" + } + }, + "notes": "Beats Alibaba's own 1T-parameter API-only Qwen3-Max with only 17B active parameters, with up to 19x faster decode at 256K context. Native multimodality and 201-language coverage are unmatched among open models." + }, + "security": { + "overall_score": 84, + "criteria": { + "prompt_injection_resistance": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Community red-team evaluations", + "url": "https://github.com/QwenLM/Qwen3.5", + "date": "2026-03-15", + "value": "Good resistance to common injection patterns; multimodal inputs add an image-based injection surface" + } + ], + "methodology": "Testing against OWASP LLM01 prompt injection patterns, including image-borne injection for multimodal inputs", + "last_verified": "2026-06-10" + }, + "jailbreak_resistance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Qwen Safety Documentation", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Safety post-training across the family; open weights mean alignment is removable downstream" + } + ], + "methodology": "Adversarial prompt testing; assessment accounts for open-weight modifiability", + "last_verified": "2026-06-10" + }, + "data_leakage_prevention": { + "score": 81, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Privacy Documentation", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-02-16", + "value": "Standard data handling on hosted endpoints; self-hosting gives complete data control" + } + ], + "methodology": "Analysis of hosted-platform policies plus the self-hosting option for full data isolation", + "last_verified": "2026-06-10" + }, + "output_safety": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Qwen3.5 Release Blog", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Multilingual safety filtering across 201 languages; refusal behavior consistent in community testing" + } + ], + "methodology": "Safety testing across harmful content categories and multiple languages on default weights", + "last_verified": "2026-06-10" + }, + "api_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-02-16", + "value": "API key authentication, HTTPS, RAM-based access control, and rate limiting on Alibaba Cloud" + } + ], + "methodology": "Review of API security features on the first-party hosted platform", + "last_verified": "2026-06-10" + } + }, + "notes": "Solid default guardrails with notably broad multilingual safety coverage. Multimodal inputs widen the attack surface; open weights shift responsibility to deployers who fine-tune." + }, + "privacy_compliance": { + "overall_score": 79, + "criteria": { + "data_residency": { + "value": "China/Singapore (Alibaba Cloud first-party); anywhere via self-hosting or third-party hosts", + "confidence": "high", + "evidence": [ + { + "source": "Alibaba Cloud Regions", + "url": "https://www.alibabacloud.com/en/global-locations", + "date": "2026-02-16", + "value": "First-party hosting on Alibaba Cloud is China-jurisdiction (with a Singapore international region); Apache-2.0 weights allow deployment in any jurisdiction" + } + ], + "methodology": "Review of hosting regions and licensing; China-jurisdiction caveat applies to Alibaba's first-party API, not self-hosted or Western-hosted deployments", + "last_verified": "2026-06-10" + }, + "training_data_optout": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio Terms", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-02-16", + "value": "Enterprise tier does not train on customer data; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of hosted-platform data usage terms", + "last_verified": "2026-06-10" + }, + "data_retention": { + "value": "Per Alibaba Cloud policy (region-dependent); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Trust Center", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-02-16", + "value": "Hosted retention follows Alibaba Cloud regional policies; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of hosted-platform retention policies; retention is deployment-dependent for open-weight models", + "last_verified": "2026-06-10" + }, + "pii_handling": { + "score": 77, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Documentation", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-02-16", + "value": "Customer responsible for PII redaction; Alibaba Cloud provides surrounding data-governance tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-06-10" + }, + "compliance_certifications": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Trust Center", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-02-16", + "value": "Alibaba Cloud holds ISO 27001/SOC reports for its infrastructure, but no HIPAA/FedRAMP path for the model service; Western-host deployments inherit those hosts' certifications" + } + ], + "methodology": "Verification of infrastructure certifications versus model-service-level compliance for Western regulated markets", + "last_verified": "2026-06-10" + }, + "zero_data_retention": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Open-weight deployment options", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-02-16", + "value": "No zero-retention guarantee on first-party hosting; self-hosting provides true zero external retention" + } + ], + "methodology": "Review of data handling across first-party API, third-party hosts, and self-hosting", + "last_verified": "2026-06-10" + } + }, + "notes": "Alibaba's first-party API is China-jurisdiction (Singapore region available), which concerns Western regulated buyers; Apache-2.0 self-hosting or Western third-party hosting fully avoids that. Alibaba Cloud's infrastructure certifications are stronger than DeepSeek's platform but still lack HIPAA/FedRAMP for the model service." + }, + "trust_transparency": { + "overall_score": 82, + "criteria": { + "explainability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Qwen3.5 Release Blog", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Hybrid thinking modes expose reasoning traces; fully inspectable when self-hosted" + } + ], + "methodology": "Evaluation of reasoning transparency and trace accessibility", + "last_verified": "2026-06-10" + }, + "hallucination_rate": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Community factuality testing", + "url": "https://github.com/QwenLM/Qwen3.5", + "date": "2026-03-20", + "value": "Moderate hallucination rate, improved over Qwen3; grounding quality on vision inputs is strong" + } + ], + "methodology": "Testing on factual QA and multimodal grounding datasets", + "last_verified": "2026-06-10" + }, + "bias_fairness": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Independent bias evaluations", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-03-01", + "value": "Broad multilingual fairness work; topic-avoidance on China-politically-sensitive subjects persists in default weights" + } + ], + "methodology": "Evaluation on bias benchmarks across languages and politically sensitive topic probes", + "last_verified": "2026-06-10" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Model behavior assessment", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-02-16", + "value": "Expresses uncertainty in thinking mode; final-answer calibration is adequate but not exceptional" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-06-10" + }, + "model_card_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-02-16", + "value": "Thorough model card with architecture details (512-expert hybrid MoE), benchmarks, usage guidance, and deployment recipes" + } + ], + "methodology": "Review of model card and technical documentation completeness", + "last_verified": "2026-06-10" + }, + "training_data_transparency": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Qwen3.5 Release Blog", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Training methodology and multilingual/multimodal data strategy described at a high level; detailed composition not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-06-10" + }, + "guardrails": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Qwen Safety Documentation", + "url": "https://github.com/QwenLM/Qwen3.5", + "date": "2026-02-16", + "value": "Multilingual safety alignment in released weights; removable by downstream fine-tuning" + } + ], + "methodology": "Analysis of built-in safety mechanisms in default weights", + "last_verified": "2026-06-10" + } + }, + "notes": "Strong open documentation and inspectable reasoning. Typical open-model gaps remain: limited training-data detail and topic-avoidance on politically sensitive subjects in default weights." + }, + "operational_excellence": { + "overall_score": 87, + "criteria": { + "api_design_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-02-16", + "value": "OpenAI-compatible API with function calling, multimodal inputs, and hybrid thinking-mode controls" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-06-10" + }, + "sdk_quality": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "QwenLM GitHub", + "url": "https://github.com/QwenLM/Qwen3.5", + "date": "2026-02-16", + "value": "Day-one support in vLLM, SGLang, and transformers; actively maintained official repos" + } + ], + "methodology": "Review of SDK and inference-framework support", + "last_verified": "2026-06-10" + }, + "versioning_policy": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Qwen Release History", + "url": "https://qwen.ai/blog?id=qwen3.5", + "date": "2026-02-16", + "value": "Fast release cadence (supersedes Qwen3 family and Qwen2.5-VL); open weights remain permanently available, softening deprecation impact" + } + ], + "methodology": "Review of release cadence and weight-availability guarantees", + "last_verified": "2026-06-10" + }, + "monitoring_observability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-02-16", + "value": "Usage dashboards and logging on Alibaba Cloud; full observability when self-hosting" + } + ], + "methodology": "Review of monitoring tools across deployment options", + "last_verified": "2026-06-10" + }, + "support_quality": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Support", + "url": "https://www.alibabacloud.com/en/contact-sales", + "date": "2026-02-16", + "value": "Alibaba Cloud offers paid enterprise support tiers; Western-market support depth lags US hyperscalers; strong community channels" + } + ], + "methodology": "Assessment of support tiers, documentation, and community responsiveness", + "last_verified": "2026-06-10" + }, + "ecosystem_maturity": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Qwen Organization", + "url": "https://huggingface.co/Qwen", + "date": "2026-03-15", + "value": "Largest open-model ecosystem by derivative count; full size ladder from 397B-A17B and 122B-A10B down to 0.8B released Feb-Mar 2026" + } + ], + "methodology": "Analysis of derivative models, third-party hosting, and tooling integrations", + "last_verified": "2026-06-10" + }, + "license_terms": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/Qwen/Qwen3.5-397B-A17B", + "date": "2026-02-16", + "value": "Apache 2.0 across the family: unrestricted commercial use with explicit patent grant" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-06-10" + } + }, + "notes": "Best-in-class open-model ecosystem: Apache 2.0 with patent grant, day-one inference-framework support, and a complete size ladder (0.8B to 397B-A17B) for matching capability to hardware. Supersedes the Qwen3 family and Qwen2.5-VL." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 92, + "notes": "Strong agentic coding that beats the 1T-parameter Qwen3-Max; 17B active params make self-hosted coding assistants economical.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "customer-support": { + "overall": 88, + "notes": "201-language coverage and fast decode make it a standout for global multilingual support; smaller variants serve high-volume tiers cheaply.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "content-creation": { + "overall": 86, + "notes": "Strong multilingual content with native image understanding for visually grounded writing.", + "alternatives": ["claude-opus-4-5", "deepseek-v4"] + }, + "data-analysis": { + "overall": 90, + "notes": "Native multimodality handles charts, tables, and documents directly; 262K context (1M on Plus) covers large datasets.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "research-assistant": { + "overall": 90, + "notes": "Multimodal document understanding plus long context suits literature and mixed-media research; 19x decode speedup keeps long-context work responsive.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "legal-compliance": { + "overall": 75, + "notes": "First-party hosting is China-jurisdiction; viable for regulated legal work only via self-hosting or certified Western hosts.", + "alternatives": ["claude-opus-4-5"] + }, + "healthcare": { + "overall": 73, + "notes": "No HIPAA path on first-party hosting; self-hosted deployment in compliant infrastructure is the only viable route.", + "alternatives": ["claude-opus-4-5"] + }, + "financial-analysis": { + "overall": 87, + "notes": "Good quantitative reasoning with native chart/table understanding; data-residency planning required for regulated workloads.", + "alternatives": ["deepseek-v4", "claude-opus-4-5"] + }, + "education": { + "overall": 91, + "notes": "201 languages, multimodal input, and a size ladder down to 0.8B make it exceptional for global and on-device education deployments.", + "alternatives": ["deepseek-v3-2", "deepseek-v4"] + }, + "creative-writing": { + "overall": 84, + "notes": "Capable multilingual creative output with visual grounding; prose distinctiveness behind dedicated creative leaders.", + "alternatives": ["claude-opus-4-5", "deepseek-v4"] + } + }, + "strengths": [ + "Beats Alibaba's own 1T-parameter API-only Qwen3-Max with just 17B active parameters (397B total)", + "Natively multimodal open model: text + vision under 'Towards Native Multimodal Agents'", + "Up to 19x faster decode than Qwen3-Max at 256K context; 262K native context (1M on hosted Qwen3.5-Plus)", + "201-language coverage, the broadest of any open model", + "Apache 2.0 license with patent grant across the entire family", + "Complete size ladder (0.8B to 397B-A17B, Feb-Mar 2026) for matching capability to hardware", + "Day-one vLLM/SGLang/transformers support and the largest open-model derivative ecosystem" + ], + "limitations": [ + "First-party Alibaba Cloud hosting is China-jurisdiction (Singapore region available); no HIPAA/FedRAMP path for the model service — self-hosting or Western hosts avoid this", + "1M context requires the hosted Qwen3.5-Plus; open weights cap at 262K", + "Topic-avoidance on politically sensitive subjects in default weights", + "Training-data composition disclosed only at a high level", + "397B total parameters still require multi-GPU infrastructure to self-host despite the sparse 17B-active design", + "Western-market enterprise support depth lags US hyperscalers" + ], + "best_for": [ + "Global multilingual products needing broad language coverage (201 languages)", + "Multimodal agents processing documents, charts, and images natively under an open license", + "Cost-efficient self-hosted inference exploiting the 17B-active sparse design", + "Organizations standardizing on one model family across sizes from edge (0.8B) to flagship", + "Long-context workloads needing fast decode at 256K tokens" + ], + "not_recommended_for": [ + "Regulated Western workloads via Alibaba's first-party API", + "Audio or video generation applications", + "Teams needing 1M context with open weights (hosted-only feature)", + "Organizations without GPU infrastructure that also cannot accept China-jurisdiction hosted APIs" + ], + "metadata": { + "pricing": { + "input": "Free weights (Apache 2.0); hosted from ~$0.40 per 1M tokens on Alibaba Cloud Model Studio", + "output": "Hosted from ~$1.20 per 1M tokens; third-party hosts vary", + "notes": "Self-hosting is infrastructure-cost-only; the 17B-active design keeps serving costs low for its capability class. Hosted Qwen3.5-Plus (1M context) priced separately.", + "last_verified": "2026-06-10" + }, + "context_window": 262144, + "max_output": 65536, + "languages": [ + "English", + "Chinese", + "Japanese", + "Korean", + "Spanish", + "French", + "German", + "Portuguese", + "Russian", + "Arabic", + "Hindi", + "Indonesian", + "Vietnamese", + "Thai", + "and 187 more (201 total)" + ], + "modalities": ["text", "image (input)", "document"], + "api_endpoint": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions", + "open_source": true, + "architecture": "Hybrid Mixture-of-Experts with 512 experts (397B total / 17B active), natively multimodal, hybrid thinking modes", + "parameters": "397B total / 17B active (flagship); family spans 0.8B to 397B-A17B", + "knowledge_cutoff": "Late 2025" + }, + "related_entities": ["deepseek-v4", "deepseek-v3-2", "kimi-k2-6", "glm-5", "claude-opus-4-5"], + "tags": [ + "open-source", + "apache-2-0", + "multimodal", + "multilingual", + "mixture-of-experts", + "long-context", + "agentic", + "chinese-provider", + "flagship" + ] +} diff --git a/lib/data.ts b/lib/data.ts index 01041f4..96e07e2 100644 --- a/lib/data.ts +++ b/lib/data.ts @@ -6,10 +6,15 @@ import type { TrustVectorEntity } from '@/framework/schema/types'; import { calculateOverallScore } from '@/framework/schema/types'; // ======================================== -// MODELS (38 total) +// MODELS (60 total) // ======================================== -// Anthropic Models (7) +// Anthropic Models (11) +import claudeFable5 from '@/data/models/claude-fable-5.json'; +import claudeOpus48 from '@/data/models/claude-opus-4-8.json'; +import claudeOpus47 from '@/data/models/claude-opus-4-7.json'; +import claudeOpus46 from '@/data/models/claude-opus-4-6.json'; +import claudeSonnet46 from '@/data/models/claude-sonnet-4-6.json'; import claudeOpus45 from '@/data/models/claude-opus-4-5.json'; import claudeSonnet45 from '@/data/models/claude-sonnet-4-5.json'; import claudeSonnet4 from '@/data/models/claude-sonnet-4.json'; @@ -17,7 +22,10 @@ import claudeOpus41 from '@/data/models/claude-opus-4-1.json'; import claudeOpus4 from '@/data/models/claude-opus-4.json'; import claudeHaiku45 from '@/data/models/claude-haiku-4-5.json'; -// OpenAI Models (17) +// OpenAI Models (20) +import gpt55 from '@/data/models/gpt-5-5.json'; +import gpt54 from '@/data/models/gpt-5-4.json'; +import gpt53Codex from '@/data/models/gpt-5-3-codex.json'; import gpt52 from '@/data/models/gpt-5-2.json'; import gpt52Codex from '@/data/models/gpt-5-2-codex.json'; import gpt51 from '@/data/models/gpt-5-1.json'; @@ -35,7 +43,10 @@ import openaiO4Mini from '@/data/models/openai-o4-mini.json'; import gptOss120b from '@/data/models/gpt-oss-120b.json'; import gptOss20b from '@/data/models/gpt-oss-20b.json'; -// Google Models (5) +// Google Models (8) +import gemini31Pro from '@/data/models/gemini-3-1-pro.json'; +import gemini35Flash from '@/data/models/gemini-3-5-flash.json'; +import gemma4 from '@/data/models/gemma-4.json'; import gemini3Pro from '@/data/models/gemini-3-pro.json'; import gemini3Flash from '@/data/models/gemini-3-flash.json'; import gemini25Pro from '@/data/models/gemini-2-5-pro.json'; @@ -49,22 +60,53 @@ import llama4Scout from '@/data/models/llama-4-scout.json'; import llama31405b from '@/data/models/llama-3-1-405b.json'; import llama3370b from '@/data/models/llama-3-3-70b.json'; -// xAI Models (1) +// xAI Models (3) +import grok43 from '@/data/models/grok-4-3.json'; +import grok41 from '@/data/models/grok-4-1.json'; import grok3Beta from '@/data/models/grok-3-beta.json'; -// DeepSeek Models (2) +// DeepSeek Models (4) +import deepseekV4 from '@/data/models/deepseek-v4.json'; +import deepseekV32 from '@/data/models/deepseek-v3-2.json'; import deepseekR1 from '@/data/models/deepseek-r1.json'; import deepseekV30324 from '@/data/models/deepseek-v3-0324.json'; -// Other Models (3) +// Other Models (10) +import qwen35 from '@/data/models/qwen3-5.json'; +import kimiK26 from '@/data/models/kimi-k2-6.json'; +import glm5 from '@/data/models/glm-5.json'; +import minimaxM2 from '@/data/models/minimax-m2.json'; +import mistralLarge3 from '@/data/models/mistral-large-3.json'; +import commandAPlus from '@/data/models/command-a-plus.json'; +import nova2Lite from '@/data/models/nova-2-lite.json'; import nemotronUltra253b from '@/data/models/nemotron-ultra-253b.json'; import qwen25Vl32b from '@/data/models/qwen2-5-vl-32b.json'; import novaPro from '@/data/models/nova-pro.json'; // ======================================== -// AGENTS (34 total) +// AGENTS (50 total) // ======================================== +// Coding & General-Purpose Agents (12, added 2026-06) +import claudeCode from '@/data/agents/claude-code.json'; +import claudeAgentSdk from '@/data/agents/claude-agent-sdk.json'; +import openaiAgentsSdk from '@/data/agents/openai-agents-sdk.json'; +import openaiCodex from '@/data/agents/openai-codex.json'; +import googleAdk from '@/data/agents/google-adk.json'; +import geminiCli from '@/data/agents/gemini-cli.json'; +import googleJules from '@/data/agents/google-jules.json'; +import githubCopilotCodingAgent from '@/data/agents/github-copilot-coding-agent.json'; +import microsoftAgentFramework from '@/data/agents/microsoft-agent-framework.json'; +import devin from '@/data/agents/devin.json'; +import cursorAgent from '@/data/agents/cursor-agent.json'; +import manus from '@/data/agents/manus.json'; + +// Open-Source Agent Frameworks (4, added 2026-06) +import smolagents from '@/data/agents/smolagents.json'; +import strandsAgents from '@/data/agents/strands-agents.json'; +import mastra from '@/data/agents/mastra.json'; +import dify from '@/data/agents/dify.json'; + // Enterprise Agents (9) import amazonLex from '@/data/agents/amazon-lex.json'; import azureBotService from '@/data/agents/azure-bot-service.json'; @@ -112,9 +154,23 @@ import autogpt from '@/data/agents/autogpt.json'; import babyagi from '@/data/agents/babyagi.json'; // ======================================== -// MCPs (34 total) +// MCPs (46 total) // ======================================== +// Top Ecosystem MCPs (12, added 2026-06) +import mcpPlaywright from '@/data/mcps/mcp-server-playwright.json'; +import mcpChromeDevtools from '@/data/mcps/mcp-server-chrome-devtools.json'; +import mcpContext7 from '@/data/mcps/mcp-server-context7.json'; +import mcpSerena from '@/data/mcps/mcp-server-serena.json'; +import mcpFigma from '@/data/mcps/mcp-server-figma.json'; +import mcpStripe from '@/data/mcps/mcp-server-stripe.json'; +import mcpVercel from '@/data/mcps/mcp-server-vercel.json'; +import mcpHuggingFace from '@/data/mcps/mcp-server-hugging-face.json'; +import mcpFirecrawl from '@/data/mcps/mcp-server-firecrawl.json'; +import mcpShadcn from '@/data/mcps/mcp-server-shadcn.json'; +import mcpApify from '@/data/mcps/mcp-server-apify.json'; +import mcpZapier from '@/data/mcps/mcp-server-zapier.json'; + // Official/Reference MCPs (5) import mcpFetch from '@/data/mcps/mcp-server-fetch.json'; import mcpGit from '@/data/mcps/mcp-server-git.json'; @@ -166,13 +222,18 @@ import mcpFilesystem from '@/data/mcps/mcp-server-filesystem.json'; import mcpMemory from '@/data/mcps/mcp-server-memory.json'; /** - * All entities in the system (106 total: 38 models + 34 agents + 34 MCPs) + * All entities in the system (156 total: 60 models + 50 agents + 46 MCPs) */ const ALL_ENTITIES: TrustVectorEntity[] = [ // ======================================== - // MODELS (38) + // MODELS (60) // ======================================== - // Anthropic (7) + // Anthropic (11) + claudeFable5, + claudeOpus48, + claudeOpus47, + claudeOpus46, + claudeSonnet46, claudeOpus45, claudeSonnet45, claudeSonnet4, @@ -180,7 +241,10 @@ const ALL_ENTITIES: TrustVectorEntity[] = [ claudeOpus4, claudeHaiku45, - // OpenAI (17) + // OpenAI (20) + gpt55, + gpt54, + gpt53Codex, gpt52, gpt52Codex, gpt51, @@ -198,7 +262,10 @@ const ALL_ENTITIES: TrustVectorEntity[] = [ gptOss120b, gptOss20b, - // Google (5) + // Google (8) + gemini31Pro, + gemini35Flash, + gemma4, gemini3Pro, gemini3Flash, gemini25Pro, @@ -212,21 +279,52 @@ const ALL_ENTITIES: TrustVectorEntity[] = [ llama31405b, llama3370b, - // xAI (1) + // xAI (3) + grok43, + grok41, grok3Beta, - // DeepSeek (2) + // DeepSeek (4) + deepseekV4, + deepseekV32, deepseekR1, deepseekV30324, - // Other (3) + // Other (10) + qwen35, + kimiK26, + glm5, + minimaxM2, + mistralLarge3, + commandAPlus, + nova2Lite, nemotronUltra253b, qwen25Vl32b, novaPro, // ======================================== - // AGENTS (34) + // AGENTS (50) // ======================================== + // Coding & General-Purpose Agents (12) + claudeCode, + claudeAgentSdk, + openaiAgentsSdk, + openaiCodex, + googleAdk, + geminiCli, + googleJules, + githubCopilotCodingAgent, + microsoftAgentFramework, + devin, + cursorAgent, + manus, + + // Open-Source Agent Frameworks — 2026 additions (4) + smolagents, + strandsAgents, + mastra, + dify, + // Enterprise (9) amazonLex, azureBotService, @@ -274,8 +372,22 @@ const ALL_ENTITIES: TrustVectorEntity[] = [ babyagi, // ======================================== - // MCPs (34) + // MCPs (46) // ======================================== + // Top Ecosystem Servers — 2026 additions (12) + mcpPlaywright, + mcpChromeDevtools, + mcpContext7, + mcpSerena, + mcpFigma, + mcpStripe, + mcpVercel, + mcpHuggingFace, + mcpFirecrawl, + mcpShadcn, + mcpApify, + mcpZapier, + // Official/Reference (5) mcpFetch, mcpGit,