From aa17882ada92fe5327388dc1ffd77694b97b622f Mon Sep 17 00:00:00 2001 From: JBAhire Date: Thu, 9 Jul 2026 22:52:43 -0700 Subject: [PATCH 1/7] =?UTF-8?q?data:=20July=202026=20registry=20refresh=20?= =?UTF-8?q?=E2=80=94=20all=20156=20evaluations=20re-verified?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every evaluation re-verified against primary sources on 2026-07-09: lifecycle statuses (retirements, deprecations, supersessions), pricing, versions, CVEs, and hosted-endpoint changes. Fixes dangling cross-references in 15 files (claude-4-opus, gpt-4-5, etc. mapped to real registry ids or dropped). Co-Authored-By: Claude Fable 5 --- data/agents/activepieces.json | 96 ++++---- data/agents/adala.json | 75 +++---- data/agents/agentgpt.json | 60 ++--- data/agents/amazon-bedrock-agents.json | 72 +++--- data/agents/amazon-lex.json | 64 +++--- data/agents/autogen.json | 64 +++--- data/agents/autogpt.json | 76 ++++--- data/agents/azure-bot-service.json | 100 +++++---- data/agents/babyagi.json | 58 ++--- data/agents/claude-agent-sdk.json | 64 +++--- data/agents/claude-code.json | 77 ++++--- data/agents/crewai.json | 111 ++++++---- data/agents/cursor-agent.json | 85 +++++--- data/agents/devin.json | 78 +++---- data/agents/dify.json | 79 ++++--- data/agents/e2b-agents.json | 69 +++--- data/agents/flowise.json | 100 +++++---- data/agents/gemini-cli.json | 115 +++++----- data/agents/github-copilot-coding-agent.json | 64 +++--- data/agents/glean-ai.json | 63 +++--- data/agents/google-adk.json | 62 +++--- data/agents/google-agent-builder.json | 71 +++--- data/agents/google-dialogflow.json | 80 ++++--- data/agents/google-jules.json | 71 +++--- data/agents/haystack.json | 77 ++++--- data/agents/ibm-watson-assistant.json | 85 +++++--- data/agents/kore-ai.json | 67 +++--- data/agents/langflow.json | 106 +++++---- data/agents/langgraph-agent.json | 81 ++++--- data/agents/llamaindex-agent.json | 75 ++++--- data/agents/make-ai.json | 92 ++++---- data/agents/manus.json | 99 +++++---- data/agents/mastra.json | 59 ++--- data/agents/memgpt.json | 79 ++++--- data/agents/microsoft-agent-framework.json | 62 +++--- data/agents/n8n-ai-agent.json | 97 +++++---- data/agents/openai-agents-sdk.json | 82 ++++--- data/agents/openai-assistants-api.json | 52 +++-- data/agents/openai-codex.json | 72 +++--- data/agents/pydantic-ai.json | 75 ++++--- data/agents/rasa.json | 92 ++++---- data/agents/relevance-ai.json | 59 ++--- data/agents/salesforce-einstein-bots.json | 64 +++--- data/agents/semantic-kernel-agent.json | 60 ++--- data/agents/sierra-ai.json | 68 +++--- data/agents/smolagents.json | 61 +++--- data/agents/strands-agents.json | 72 +++--- data/agents/superagi.json | 103 +++++---- data/agents/swarm.json | 70 +++--- data/agents/zapier-ai.json | 98 +++++---- data/mcps/mcp-server-apify.json | 85 ++++---- data/mcps/mcp-server-atlassian.json | 103 ++++++--- data/mcps/mcp-server-aws.json | 205 +++++++++--------- data/mcps/mcp-server-azure.json | 178 +++++++-------- data/mcps/mcp-server-brave-search.json | 58 ++--- data/mcps/mcp-server-calendar.json | 66 +++--- data/mcps/mcp-server-chrome-devtools.json | 67 +++--- data/mcps/mcp-server-cloudflare.json | 197 ++++++++--------- data/mcps/mcp-server-context7.json | 65 +++--- data/mcps/mcp-server-datadog.json | 131 +++++++---- data/mcps/mcp-server-docker.json | 94 ++++---- data/mcps/mcp-server-elasticsearch.json | 198 +++++++++-------- data/mcps/mcp-server-everything.json | 62 +++--- data/mcps/mcp-server-fetch.json | 66 +++--- data/mcps/mcp-server-figma.json | 65 +++--- data/mcps/mcp-server-filesystem.json | 72 +++--- data/mcps/mcp-server-firecrawl.json | 111 +++++----- data/mcps/mcp-server-git.json | 82 ++++--- data/mcps/mcp-server-github.json | 79 ++++--- data/mcps/mcp-server-gitlab.json | 65 +++--- data/mcps/mcp-server-gmail.json | 66 +++--- data/mcps/mcp-server-google-drive.json | 63 +++--- data/mcps/mcp-server-hugging-face.json | 58 ++--- data/mcps/mcp-server-kubernetes.json | 89 +++++--- data/mcps/mcp-server-linear.json | 121 +++++++---- data/mcps/mcp-server-memory.json | 64 +++--- data/mcps/mcp-server-mongodb.json | 201 ++++++++--------- data/mcps/mcp-server-notion.json | 75 +++---- data/mcps/mcp-server-perplexity.json | 99 +++++---- data/mcps/mcp-server-playwright.json | 67 +++--- data/mcps/mcp-server-postgres.json | 62 +++--- data/mcps/mcp-server-puppeteer.json | 60 ++--- data/mcps/mcp-server-redis.json | 88 ++++---- data/mcps/mcp-server-s3.json | 134 ++++++------ data/mcps/mcp-server-sentry.json | 124 +++++++---- data/mcps/mcp-server-sequential-thinking.json | 62 +++--- data/mcps/mcp-server-serena.json | 70 +++--- data/mcps/mcp-server-shadcn.json | 85 ++++---- data/mcps/mcp-server-slack.json | 63 +++--- data/mcps/mcp-server-sqlite.json | 52 ++--- data/mcps/mcp-server-stripe.json | 104 ++++----- data/mcps/mcp-server-supabase.json | 142 ++++++------ data/mcps/mcp-server-tavily.json | 110 +++++----- data/mcps/mcp-server-time.json | 62 +++--- data/mcps/mcp-server-vercel.json | 150 ++++++------- data/mcps/mcp-server-zapier.json | 110 +++++----- data/models/claude-fable-5.json | 136 +++++++----- data/models/claude-haiku-4-5.json | 132 +++++------ data/models/claude-opus-4-1.json | 134 ++++++------ data/models/claude-opus-4-5.json | 87 ++++---- data/models/claude-opus-4-6.json | 86 ++++---- data/models/claude-opus-4-7.json | 86 ++++---- data/models/claude-opus-4-8.json | 89 ++++---- data/models/claude-opus-4.json | 125 ++++++----- data/models/claude-sonnet-4-5.json | 137 ++++++------ data/models/claude-sonnet-4-6.json | 113 +++++----- data/models/claude-sonnet-4.json | 122 ++++++----- data/models/command-a-plus.json | 82 +++---- data/models/deepseek-r1.json | 111 +++++----- data/models/deepseek-v3-0324.json | 91 ++++---- data/models/deepseek-v3-2.json | 72 +++--- data/models/deepseek-v4.json | 86 ++++---- data/models/gemini-2-0-flash.json | 100 +++++---- data/models/gemini-2-5-pro.json | 113 +++++----- data/models/gemini-3-1-pro.json | 131 +++++------ data/models/gemini-3-5-flash.json | 84 +++---- data/models/gemini-3-flash.json | 95 ++++---- data/models/gemini-3-pro.json | 94 ++++---- data/models/gemma-3-27b.json | 109 +++++----- data/models/gemma-4.json | 68 +++--- data/models/glm-5.json | 86 ++++---- data/models/gpt-4-1-mini.json | 105 +++++---- data/models/gpt-4-1-nano.json | 117 +++++----- data/models/gpt-4-1.json | 103 +++++---- data/models/gpt-4o-mini.json | 83 +++---- data/models/gpt-4o.json | 85 ++++---- data/models/gpt-5-1.json | 106 ++++----- data/models/gpt-5-2-codex.json | 76 +++---- data/models/gpt-5-2.json | 84 +++---- data/models/gpt-5-3-codex.json | 93 ++++---- data/models/gpt-5-4.json | 80 +++---- data/models/gpt-5-5.json | 91 ++++---- data/models/gpt-5.json | 117 +++++----- data/models/gpt-oss-120b.json | 98 +++++---- data/models/gpt-oss-20b.json | 104 +++++---- data/models/grok-3-beta.json | 82 +++---- data/models/grok-4-1.json | 82 +++---- data/models/grok-4-3.json | 115 +++++----- data/models/kimi-k2-6.json | 86 ++++---- data/models/llama-3-1-405b.json | 78 ++++--- data/models/llama-3-3-70b.json | 91 ++++---- data/models/llama-4-behemoth.json | 78 ++++--- data/models/llama-4-maverick.json | 105 ++++----- data/models/llama-4-scout.json | 152 +++++++------ data/models/minimax-m2.json | 86 ++++---- data/models/mistral-large-3.json | 78 +++---- data/models/nemotron-ultra-253b.json | 106 ++++----- data/models/nova-2-lite.json | 74 +++---- data/models/nova-pro.json | 78 +++---- data/models/openai-o1-mini.json | 86 ++++---- data/models/openai-o1.json | 103 +++++---- data/models/openai-o3-mini.json | 100 +++++---- data/models/openai-o3.json | 90 ++++---- data/models/openai-o4-mini.json | 106 +++++---- data/models/qwen2-5-vl-32b.json | 98 +++++---- data/models/qwen3-5.json | 87 ++++---- 156 files changed, 7633 insertions(+), 6530 deletions(-) diff --git a/data/agents/activepieces.json b/data/agents/activepieces.json index d3f1138..f5ca87a 100644 --- a/data/agents/activepieces.json +++ b/data/agents/activepieces.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Activepieces", "provider": "Activepieces", - "version": "0.x", - "last_evaluated": "2025-11-09", + "version": "0.86.2", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source no-code business automation platform with AI capabilities. Self-hostable alternative to Zapier with visual workflow builder, 200+ app integrations, and LLM integration for intelligent automation.", + "description": "Open-source no-code business automation platform with AI agent capabilities. Self-hostable alternative to Zapier with visual workflow builder, 700+ app integrations (each also exposed as an MCP server for AI agents/LLM tools), and LLM integration for intelligent automation.", "website": "https://www.activepieces.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Workflow reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ai_integration": { "score": 79, @@ -38,7 +38,7 @@ } ], "methodology": "AI capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "app_integrations": { "score": 82, @@ -47,12 +47,12 @@ { "source": "Pieces Catalog", "url": "https://www.activepieces.com/pieces", - "date": "2024-10-01", - "value": "200+ app integrations (pieces) with active development" + "date": "2026-07-09", + "value": "700+ app integrations (pieces) with active development; ~400 exposed as MCP servers" } ], "methodology": "Integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 76, @@ -66,7 +66,7 @@ } ], "methodology": "Error recovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "performance": { "score": 74, @@ -80,7 +80,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Variable (self-hosted)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "authentication": { "score": 75, @@ -127,7 +127,7 @@ } ], "methodology": "Authentication testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_storage": { "score": 78, @@ -141,7 +141,7 @@ } ], "methodology": "Credential security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 88, @@ -150,12 +150,12 @@ { "source": "GitHub", "url": "https://github.com/activepieces/activepieces", - "date": "2024-10-20", - "value": "MIT license, 9k+ stars, open source community" + "date": "2026-07-09", + "value": "MIT license, 23k+ stars, open source community" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_security": { "score": 70, @@ -169,7 +169,7 @@ } ], "methodology": "API security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 81, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 93, @@ -216,7 +216,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_processing": { "score": 78, @@ -230,7 +230,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "no_telemetry": { "score": 85, @@ -244,7 +244,7 @@ } ], "methodology": "Telemetry assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "visual_builder": { "score": 87, @@ -277,7 +277,7 @@ } ], "methodology": "UI/UX assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 90, @@ -286,12 +286,12 @@ { "source": "GitHub", "url": "https://github.com/activepieces/activepieces", - "date": "2024-10-20", - "value": "MIT license, 9k+ stars, active development" + "date": "2026-07-09", + "value": "MIT license, 23k+ stars, active development (0.86.2 released July 8, 2026)" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_logs": { "score": 78, @@ -305,7 +305,7 @@ } ], "methodology": "Traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 76, @@ -319,7 +319,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 70, @@ -352,7 +352,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 92, @@ -361,12 +361,12 @@ { "source": "Pricing", "url": "https://www.activepieces.com/pricing", - "date": "2024-10-01", - "value": "Free MIT open source, cloud from $250/month" + "date": "2026-07-09", + "value": "Free MIT open source (Community Edition); Cloud Standard: 10 free active flows then $5/active flow/month with unlimited runs; Ultimate: custom annual contract" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 68, @@ -380,7 +380,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "piece_ecosystem": { "score": 76, @@ -389,12 +389,12 @@ { "source": "Pieces", "url": "https://www.activepieces.com/pieces", - "date": "2024-10-01", - "value": "200+ pieces (integrations), growing ecosystem" + "date": "2026-07-09", + "value": "700+ pieces (integrations), fast-growing ecosystem with MCP server support" } ], "methodology": "Integration ecosystem assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 72, @@ -408,7 +408,7 @@ } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -466,8 +466,8 @@ "No vendor lock-in with full data control", "Clean visual no-code interface for business users", "Free to use with only infrastructure costs", - "200+ app integrations with active development", - "Growing community and active GitHub development" + "700+ app integrations with active development, each usable as an MCP server", + "Growing community and active GitHub development (23k+ stars)" ], "limitations": [ "Smaller integration ecosystem than commercial alternatives", @@ -489,16 +489,18 @@ ], "deployment_type": "Self-hosted (Docker) or Activepieces Cloud", "tool_support": [ - "200+ app pieces", + "700+ app pieces", "Custom pieces", - "Webhooks" + "Webhooks", + "MCP servers" ], - "pricing_model": "Free open source, Cloud from $250/month", - "github_stars": "9000+", + "pricing_model": "Free open source (Community Edition); Cloud Standard: flow-based pricing ($5/active flow/month after 10 free, unlimited runs); Ultimate: custom", + "github_stars": "23000+", "first_release": "2022", + "latest_version": "0.86.2 (July 8, 2026)", "database": "PostgreSQL", "queue_system": "Redis (optional for scaling)", - "pricing": "Free self-hosted (MIT license), Cloud: $1 per 1,000 tasks. 300+ integrations, 280+ MCP servers" + "pricing": "Free self-hosted (MIT license); Cloud Standard: 10 free active flows, $5/active flow/month thereafter, unlimited runs (verified 2026-07-09). 700+ integrations, ~400 MCP servers" }, "tags": [ "automation", diff --git a/data/agents/adala.json b/data/agents/adala.json index 55d0019..e6721db 100644 --- a/data/agents/adala.json +++ b/data/agents/adala.json @@ -4,7 +4,7 @@ "name": "Adala", "provider": "HumanSignal", "version": "0.x", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Autonomous data labeling agent framework for creating self-improving AI systems. Combines LLMs with ground truth learning to automate and improve data annotation tasks, enabling continuous learning loops.", "website": "https://github.com/HumanSignal/Adala", @@ -24,7 +24,7 @@ } ], "methodology": "Labeling accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_improvement": { "score": 80, @@ -38,7 +38,7 @@ } ], "methodology": "Learning capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "skill_acquisition": { "score": 78, @@ -52,7 +52,7 @@ } ], "methodology": "Skill capability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "batch_processing": { "score": 76, @@ -66,7 +66,7 @@ } ], "methodology": "Batch processing testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ground_truth_learning": { "score": 84, @@ -80,7 +80,7 @@ } ], "methodology": "Learning effectiveness testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Variable (batch processing)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Data security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_hosting": { "score": 85, @@ -127,7 +127,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 88, @@ -136,12 +136,12 @@ { "source": "GitHub", "url": "https://github.com/HumanSignal/Adala", - "date": "2024-10-20", - "value": "Apache 2.0 license, 1k+ stars, transparent code" + "date": "2026-07-09", + "value": "Apache 2.0 license, 1,610 stars, transparent code; repo last pushed June 19, 2026" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_security": { "score": 68, @@ -155,7 +155,7 @@ } ], "methodology": "LLM security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 65, @@ -169,7 +169,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 75, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "training_data_privacy": { "score": 72, @@ -230,7 +230,7 @@ } ], "methodology": "Training data privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 70, @@ -244,7 +244,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "learning_transparency": { "score": 82, @@ -277,7 +277,7 @@ } ], "methodology": "Transparency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 88, @@ -286,12 +286,12 @@ { "source": "GitHub", "url": "https://github.com/HumanSignal/Adala", - "date": "2024-10-20", - "value": "Apache 2.0, developed by HumanSignal (Label Studio team)" + "date": "2026-07-09", + "value": "Apache 2.0, developed by HumanSignal (Label Studio team); recent commit history (June 2026) is dominated by periodic maintenance and dependency updates rather than feature development" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "skill_visibility": { "score": 76, @@ -305,7 +305,7 @@ } ], "methodology": "Explainability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 72, @@ -319,7 +319,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "label_studio_integration": { "score": 85, @@ -352,7 +352,7 @@ } ], "methodology": "Integration assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 72, @@ -366,7 +366,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -380,7 +380,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 70, @@ -394,7 +394,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 70, @@ -403,12 +403,12 @@ { "source": "Maturity", "url": "https://github.com/HumanSignal/Adala", - "date": "2024-10-01", - "value": "Active development, production use requires careful setup" + "date": "2026-07-09", + "value": "Repo not archived and receives periodic maintenance commits (last push June 19, 2026), but activity clusters suggest automated dependency updates; 168 open issues and still 0.x - production use requires careful setup" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -475,7 +475,8 @@ "Smaller community and ecosystem than general frameworks", "Limited production features and documentation", "Best suited for batch processing, not real-time inference", - "Requires expertise in data labeling workflows" + "Requires expertise in data labeling workflows", + "Development pace has slowed to periodic maintenance (mostly dependency updates as of mid-2026); still pre-1.0 with a large open-issue backlog" ], "metadata": { "license": "Apache 2.0", @@ -495,12 +496,12 @@ "Custom skills" ], "pricing_model": "Free open source", - "github_stars": "1289+", + "github_stars": "1610+", "first_release": "2024", "parent_project": "HumanSignal (Label Studio)", "use_case_focus": "Autonomous data labeling and annotation", "pricing": "Free (Apache-2.0 license)", - "updated": "November 6, 2025" + "updated": "June 19, 2026 (last repository push; verified 2026-07-09)" }, "tags": [ "data-labeling", diff --git a/data/agents/agentgpt.json b/data/agents/agentgpt.json index 2b66ab4..26ebe8d 100644 --- a/data/agents/agentgpt.json +++ b/data/agents/agentgpt.json @@ -4,7 +4,7 @@ "name": "AgentGPT", "provider": "Reworkd", "version": "Platform", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "DISCONTINUED: the AgentGPT repository was archived on 2026-01-28 (last release v1.0.0, Nov 2023) and the hosted site is frozen. Formerly a browser-based autonomous AI agent platform that let users create goal-oriented agents which break down objectives and execute tasks without continuous human intervention. Not recommended for new use.", "website": "https://agentgpt.reworkd.ai/", @@ -24,7 +24,7 @@ } ], "methodology": "Autonomous task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "task_planning": { "score": 75, @@ -38,7 +38,7 @@ } ], "methodology": "Planning capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "browser_execution": { "score": 78, @@ -52,7 +52,7 @@ } ], "methodology": "Browser capability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "iteration_limits": { "score": 70, @@ -66,7 +66,7 @@ } ], "methodology": "Execution limits testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 71, @@ -80,7 +80,7 @@ } ], "methodology": "Reliability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "3-20s per iteration", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "API key security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_privacy": { "score": 72, @@ -127,7 +127,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "browser_security": { "score": 75, @@ -141,7 +141,7 @@ } ], "methodology": "Browser security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 85, @@ -155,7 +155,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "action_sandboxing": { "score": 55, @@ -169,7 +169,7 @@ } ], "methodology": "Sandboxing assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 68, @@ -202,7 +202,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_hosting": { "score": 78, @@ -216,7 +216,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 70, @@ -230,7 +230,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "session_privacy": { "score": 72, @@ -244,7 +244,7 @@ } ], "methodology": "Session privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "web_interface": { "score": 85, @@ -277,7 +277,7 @@ } ], "methodology": "UI/UX assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 88, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_visibility": { "score": 78, @@ -305,7 +305,7 @@ } ], "methodology": "Traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 55, @@ -325,7 +325,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Score reduced: repository archived, no further community development" } } @@ -345,7 +345,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 62, @@ -359,7 +359,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 75, @@ -373,7 +373,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 65, @@ -387,7 +387,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 40, @@ -402,12 +402,12 @@ { "source": "GitHub Repository Status", "url": "https://github.com/reworkd/AgentGPT", - "date": "2026-06-10", - "value": "Repository archived 2026-01-28; last release v1.0.0 (Nov 2023); hosted site frozen" + "date": "2026-07-09", + "value": "Re-verified 2026-07-09: repository remains archived (since 2026-01-28); last release v1.0.0 (Nov 2023); hosted site frozen" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Score reduced: project discontinued and repository archived on 2026-01-28" }, "reliability": { @@ -422,7 +422,7 @@ } ], "methodology": "Reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } diff --git a/data/agents/amazon-bedrock-agents.json b/data/agents/amazon-bedrock-agents.json index d6bdead..5c12c9c 100644 --- a/data/agents/amazon-bedrock-agents.json +++ b/data/agents/amazon-bedrock-agents.json @@ -4,9 +4,9 @@ "name": "Amazon Bedrock Agents", "provider": "Amazon Web Services", "version": "2024", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Fully managed AWS service for building and deploying generative AI agents with orchestration, memory, knowledge bases, and action groups. Note: AWS's strategic agent runtime is now Bedrock AgentCore (GA 2025-10-13; framework-agnostic, 8-hour sessions, session isolation) paired with the open-source Strands Agents SDK; evaluate AgentCore for new builds.", + "description": "Fully managed AWS service for building and deploying generative AI agents with orchestration, memory, knowledge bases, and action groups. Note: AWS's strategic agent runtime is now Bedrock AgentCore (GA 2025-10-13; framework-agnostic and model-agnostic - works with models in or outside Bedrock including OpenAI and Gemini; 8-hour sessions, session isolation; expanded at re:Invent 2025) paired with the open-source Strands Agents SDK; evaluate AgentCore for new builds.", "website": "https://aws.amazon.com/bedrock/agents/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on managed service SLA and model performance", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 86, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 92, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 89, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Medium (2-6s typical)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 96, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 88, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 95, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "enterprise_security": { "score": 98, @@ -169,7 +169,7 @@ } ], "methodology": "Enterprise security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 95, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 92, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "regional_deployment": { "score": 93, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 87, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "managed_service_sla": { "score": 92, @@ -291,7 +291,7 @@ } ], "methodology": "SLA review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "proprietary_service": { "score": 65, @@ -305,7 +305,7 @@ } ], "methodology": "Transparency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 96, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 85, @@ -352,7 +352,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09", + "last_verified": "2026-07-09", "notes": "Costs can accumulate with heavy use" }, "monitoring_capabilities": { @@ -367,7 +367,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 96, @@ -384,10 +384,16 @@ "url": "https://aws.amazon.com/about-aws/whats-new/2025/10/amazon-bedrock-agentcore-available/", "date": "2026-06-10", "value": "AWS's strategic agent runtime is now Bedrock AgentCore (GA 2025-10-13): framework-agnostic, 8-hour sessions, session isolation; complemented by the open-source Strands Agents SDK" + }, + { + "source": "Amazon Bedrock AgentCore", + "url": "https://aws.amazon.com/bedrock/agentcore/", + "date": "2026-07-09", + "value": "AgentCore expanded at re:Invent 2025 (agent registry, managed multi-agent orchestration, operational tooling) and is model-agnostic, working with any foundation model in or outside Bedrock" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -451,7 +457,7 @@ "limitations": [ "AWS vendor lock-in with proprietary service", "Higher costs compared to self-hosted solutions", - "Limited to AWS Bedrock foundation models", + "Classic Bedrock Agents limited to Bedrock foundation models (AgentCore, by contrast, is model-agnostic)", "Less flexibility than code-based frameworks", "Requires AWS expertise for optimal configuration", "Not open source, limited customization of core orchestration", @@ -461,9 +467,15 @@ "license": "Proprietary (AWS)", "supported_models": [ "Claude (Anthropic)", - "Titan (Amazon)", + "Nova and Titan (Amazon)", "Llama (Meta)", - "Command (Cohere)" + "Mistral", + "Command (Cohere)", + "AI21", + "DeepSeek", + "Gemma (Google)", + "OpenAI open-weight models", + "Qwen" ], "programming_languages": [ "AWS SDK (Python, JavaScript, Java, .NET, etc.)", diff --git a/data/agents/amazon-lex.json b/data/agents/amazon-lex.json index d66843b..ef060b4 100644 --- a/data/agents/amazon-lex.json +++ b/data/agents/amazon-lex.json @@ -4,9 +4,9 @@ "name": "Amazon Lex", "provider": "Amazon Web Services", "version": "V2", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "AWS managed conversational AI service for building chatbots and voice assistants with automatic speech recognition (ASR) and natural language understanding (NLU). Integrates natively with AWS services and enterprise systems.", + "description": "AWS managed conversational AI service for building chatbots and voice assistants with automatic speech recognition (ASR) and natural language understanding (NLU). Integrates natively with AWS services and enterprise systems. Lex V2 remains an active service (V1 was discontinued September 15, 2025), but AWS's strategic direction for generative/LLM-based agents is Amazon Bedrock Agents and Bedrock AgentCore; Lex is positioned for intent-based bots and Amazon Connect contact-center flows.", "website": "https://aws.amazon.com/lex/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on enterprise deployment metrics", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "intent_recognition": { "score": 91, @@ -38,7 +38,7 @@ } ], "methodology": "Intent classification testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_language_support": { "score": 87, @@ -52,7 +52,7 @@ } ], "methodology": "Multi-language testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "voice_recognition": { "score": 92, @@ -66,7 +66,7 @@ } ], "methodology": "Voice accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "context_retention": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Conversation state testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "200-500ms average", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 94, @@ -127,7 +127,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "vpc_isolation": { "score": 93, @@ -141,7 +141,7 @@ } ], "methodology": "Network isolation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 90, @@ -155,7 +155,7 @@ } ], "methodology": "Audit capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 95, @@ -169,7 +169,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 93, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "hipaa_compliance": { "score": 94, @@ -216,7 +216,7 @@ } ], "methodology": "Healthcare compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -230,7 +230,7 @@ } ], "methodology": "PII protection testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_residency": { "score": 92, @@ -244,7 +244,7 @@ } ], "methodology": "Data residency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversation_analytics": { "score": 88, @@ -277,7 +277,7 @@ } ], "methodology": "Analytics capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "testing_tools": { "score": 85, @@ -291,7 +291,7 @@ } ], "methodology": "Testing features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "version_control": { "score": 82, @@ -305,7 +305,7 @@ } ], "methodology": "Version management assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 94, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 87, @@ -349,10 +349,16 @@ "url": "https://aws.amazon.com/lex/pricing/", "date": "2024-10-01", "value": "Pay per request pricing, $0.00075 per text, $0.004 per voice" + }, + { + "source": "Amazon Lex pricing page (re-verified)", + "url": "https://aws.amazon.com/lex/pricing/", + "date": "2026-07-09", + "value": "Confirmed current: $0.00075 per text request, $0.004 per speech request; Automated Chatbot Designer $0.50/minute of training; new AWS customers get up to $200 Free Tier credits (from July 15, 2025; valid 6 months, credits expire within 12 months)" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 91, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "channel_integrations": { "score": 89, @@ -380,7 +386,7 @@ } ], "methodology": "Integration options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sla_guarantee": { "score": 95, @@ -394,7 +400,7 @@ } ], "methodology": "SLA review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -446,7 +452,7 @@ "KMS encryption", "CloudTrail logging" ], - "pricing": "Pay-per-request: Free tier (first year: 10k text + 5k speech/month), then $0.00075-0.004 per request. $200 free tier credits for 6 months (starting July 2025)", + "pricing": "Pay-per-request: $0.00075 per text request, $0.004 per speech request (re-verified 2026-07-09). New AWS customers receive up to $200 Free Tier credits (from July 15, 2025; 6-month tier, credits expire within 12 months). Automated Chatbot Designer: $0.50 per minute of training", "version": "V2", "deprecation": "V1 discontinued September 15, 2025" }, diff --git a/data/agents/autogen.json b/data/agents/autogen.json index d894548..9e2e8f6 100644 --- a/data/agents/autogen.json +++ b/data/agents/autogen.json @@ -4,9 +4,9 @@ "name": "Microsoft AutoGen", "provider": "Microsoft Research", "version": "0.4", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "MAINTENANCE MODE: AutoGen now receives bug/security fixes only and is superseded by the Microsoft Agent Framework (1.0 GA on 2026-04-03), the recommended migration path. AutoGen is a multi-agent conversation framework for LLM applications with conversable agents combining LLMs, human input, and tools across complex workflows.", + "description": "MAINTENANCE MODE: AutoGen now receives bug/security fixes only, is community-managed going forward, and is superseded by the Microsoft Agent Framework (1.0 GA on 2026-04-03), the recommended migration path. AutoGen is a multi-agent conversation framework for LLM applications with conversable agents combining LLMs, human input, and tools across complex workflows.", "website": "https://microsoft.github.io/autogen/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on research benchmarks and model performance", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversation_quality": { "score": 89, @@ -94,7 +94,7 @@ } ], "methodology": "Conversation quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 78, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 82, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 85, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 93, @@ -169,7 +169,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 83, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 78, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 92, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 85, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 83, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 93, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "research_foundation": { "score": 95, @@ -305,7 +305,7 @@ } ], "methodology": "Academic backing assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 87, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 89, @@ -352,7 +352,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 82, @@ -366,7 +366,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 88, @@ -383,10 +383,16 @@ "url": "https://devblogs.microsoft.com/semantic-kernel/migrate-your-semantic-kernel-and-autogen-projects-to-microsoft-agent-framework-release-candidate/", "date": "2026-06-10", "value": "AutoGen is in maintenance mode (bug/security fixes only); Microsoft Agent Framework 1.0 reached GA on 2026-04-03 as the successor" + }, + { + "source": "AutoGen GitHub Repository", + "url": "https://github.com/microsoft/autogen", + "date": "2026-07-09", + "value": "Repo confirms maintenance mode: no new features or enhancements, community-managed going forward; ~59,600 stars; last substantive push 2026-04-15" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -474,12 +480,12 @@ "Function calling", "Custom tools" ], - "github_stars": "50400+", + "github_stars": "59600+", "first_release": "2023", "pricing": "Free (Apache 2.0) - Costs only from LLM API usage", "python_requirement": "Python 3.10+", "contributors": "559+", - "transition_notice": "Microsoft Agent Framework is the recommended path forward; AutoGen receives maintenance and critical patches only" + "transition_notice": "Microsoft Agent Framework is the recommended path forward; AutoGen receives maintenance and critical patches only and is community-managed going forward" }, "related_entities": [ "microsoft-agent-framework" diff --git a/data/agents/autogpt.json b/data/agents/autogpt.json index 6d0070e..693cf60 100644 --- a/data/agents/autogpt.json +++ b/data/agents/autogpt.json @@ -3,10 +3,10 @@ "type": "agent", "name": "AutoGPT", "provider": "Significant Gravitas", - "version": "0.6.36", - "last_evaluated": "2025-11-09", + "version": "0.6.61", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Autonomous AI agent that attempts to achieve user-defined goals by breaking them into sub-tasks, using internet access, memory management, and file operations. Pioneered the autonomous agent paradigm.", + "description": "Autonomous AI agent project that pioneered the autonomous agent paradigm. Development has shifted from the classic self-hosted agent to the AutoGPT Platform (still in beta as of mid-2026), a visual agent builder with workflow automation, an expanding plugin/block ecosystem, and cloud-hosted deployment alongside self-hosting. The repository remains one of the most starred open-source projects (185k+ stars) with active development.", "website": "https://docs.agpt.co/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on community feedback and testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 72, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 70, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 75, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 62, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "autonomy_level": { "score": 78, @@ -94,7 +94,7 @@ } ], "methodology": "Autonomy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 65, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 60, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 68, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -164,12 +164,12 @@ { "source": "GitHub Repository", "url": "https://github.com/Significant-Gravitas/AutoGPT", - "date": "2024-10-15", - "value": "MIT licensed, 165k+ stars, fully transparent code" + "date": "2026-07-09", + "value": "MIT licensed, 185k+ stars, fully transparent code" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 68, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 65, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 78, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 74, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 76, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 92, @@ -286,12 +286,12 @@ { "source": "GitHub Repository", "url": "https://github.com/Significant-Gravitas/AutoGPT", - "date": "2024-10-15", - "value": "MIT licensed, 165k+ stars, pioneering autonomous agent project" + "date": "2026-07-09", + "value": "MIT licensed, 185k+ stars, pioneering autonomous agent project" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_activity": { "score": 65, @@ -302,10 +302,17 @@ "url": "https://github.com/Significant-Gravitas/AutoGPT", "date": "2024-10-15", "value": "Activity has decreased from peak, focus shifting to AutoGPT Platform" + }, + { + "source": "GitHub Repository / Releases", + "url": "https://github.com/Significant-Gravitas/AutoGPT/releases", + "date": "2026-07-09", + "value": "Development active on AutoGPT Platform: repo pushed 2026-07-09, 185k+ stars, steady platform beta releases through 2026 (v0.6.61, May 2026)" } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09", + "notes": "Focus is now the AutoGPT Platform rather than the classic agent; platform development is active" } } }, @@ -324,7 +331,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 62, @@ -338,7 +345,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 60, @@ -352,7 +359,7 @@ } ], "methodology": "Cost analysis", - "last_verified": "2025-11-09", + "last_verified": "2026-07-09", "notes": "Can rack up significant API costs in continuous mode" }, "monitoring_capabilities": { @@ -367,7 +374,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 58, @@ -381,7 +388,7 @@ } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -419,9 +426,10 @@ "Code execution", "Plugins" ], - "github_stars": "180000+", + "github_stars": "185000+", "first_release": "2023", - "latest_release": "autogpt-platform-beta-v0.6.36 (November 2025)" + "latest_release": "autogpt-platform-beta-v0.6.61 (May 2026)", + "platform_status": "AutoGPT Platform in beta (visual agent builder, cloud or self-hosted); classic agent no longer the development focus" }, "use_case_ratings": { "customer-support": { diff --git a/data/agents/azure-bot-service.json b/data/agents/azure-bot-service.json index 6be7a87..60e24cb 100644 --- a/data/agents/azure-bot-service.json +++ b/data/agents/azure-bot-service.json @@ -4,9 +4,9 @@ "name": "Azure Bot Service", "provider": "Microsoft", "version": "v4", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Microsoft's enterprise bot development platform integrated with Azure AI services. Provides comprehensive tools for building, testing, deploying, and managing intelligent bots across multiple channels with advanced AI capabilities.", + "description": "LEGACY/RETIRING: the underlying Bot Framework SDK is retired — final long-term support ended December 31, 2025 (no further updates; Azure portal support tickets no longer serviced), and new multi-tenant bot creation was deprecated after July 31, 2025. Microsoft directs new projects to Copilot Studio or the Microsoft 365 Agents SDK (GA; C#, JavaScript, Python). Existing bots continue to function, and Azure Bot Service channels remain in use (including by Copilot Studio).", "website": "https://azure.microsoft.com/en-us/services/bot-services/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on enterprise deployment metrics", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "language_understanding": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "NLU testing and benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_turn_conversations": { "score": 89, @@ -52,7 +52,7 @@ } ], "methodology": "Complex conversation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "adaptive_cards_support": { "score": 92, @@ -66,7 +66,7 @@ } ], "methodology": "UI capability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "state_management": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "State persistence testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "300-800ms average", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Authentication testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_encryption": { "score": 93, @@ -127,7 +127,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 92, @@ -141,7 +141,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "network_security": { "score": 90, @@ -155,7 +155,7 @@ } ], "methodology": "Network security testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 93, @@ -169,7 +169,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 92, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "hipaa_compliance": { "score": 91, @@ -216,7 +216,7 @@ } ], "methodology": "Healthcare compliance review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversation_logging": { "score": 87, @@ -230,7 +230,7 @@ } ], "methodology": "Privacy features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_residency": { "score": 91, @@ -244,7 +244,7 @@ } ], "methodology": "Data residency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "testing_tools": { "score": 88, @@ -277,7 +277,7 @@ } ], "methodology": "Testing capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "analytics_insights": { "score": 84, @@ -291,7 +291,7 @@ } ], "methodology": "Analytics features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_sdk": { "score": 80, @@ -302,10 +302,17 @@ "url": "https://github.com/Microsoft/botframework-sdk", "date": "2024-10-15", "value": "Open source SDK available on GitHub" + }, + { + "source": "Bot Framework SDK retirement notice", + "url": "https://github.com/microsoft/botframework-sdk", + "date": "2026-07-09", + "value": "Bot Framework SDK retired; final long-term support ended December 31, 2025; no further updates and no Azure portal support tickets; Microsoft directs developers to the Microsoft 365 Agents SDK and Copilot Studio" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09", + "notes": "SDK repository remains public but is retired; no further maintenance" } } }, @@ -321,10 +328,16 @@ "url": "https://docs.microsoft.com/en-us/azure/bot-service/bot-service-overview", "date": "2024-10-01", "value": "SDKs for C#, JavaScript, Python, Java with rich tooling" + }, + { + "source": "Microsoft 365 Agents SDK migration guidance", + "url": "https://learn.microsoft.com/en-us/microsoft-365/agents-sdk/bf-migration-guidance", + "date": "2026-07-09", + "value": "Microsoft 365 Agents SDK (GA; C#, JavaScript, Python) is the evolution of the Bot Framework SDK; official migration guidance published for existing bots" } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 93, @@ -338,7 +351,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 85, @@ -352,7 +365,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 90, @@ -366,7 +379,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "channel_support": { "score": 94, @@ -380,7 +393,7 @@ } ], "methodology": "Channel integration assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "development_tools": { "score": 87, @@ -394,7 +407,7 @@ } ], "methodology": "Development tooling assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -403,11 +416,12 @@ "Deep integration with Microsoft ecosystem (Teams, Office 365, Dynamics)", "Comprehensive channel support (15+ platforms) with adaptive cards", "Enterprise-grade security and compliance (HIPAA, SOC 2, FedRAMP)", - "Visual development tools (Bot Framework Composer) for low-code scenarios", - "Open-source SDK with strong community and extensive documentation", - "Azure AI Services integration (LUIS, QnA Maker, Azure OpenAI)" + "Visual development tools (Bot Framework Composer, now legacy) for low-code scenarios", + "Open-source SDK with strong community and extensive documentation (SDK now retired)", + "Azure AI Services integration (CLU, Azure OpenAI; legacy LUIS and QnA Maker have been retired)" ], "limitations": [ + "Bot Framework SDK retired: final long-term support ended December 31, 2025 (no product/feature updates, no Azure portal support tickets); new multi-tenant bot creation deprecated after July 31, 2025; new bot projects should start on Microsoft Copilot Studio or the Microsoft 365 Agents SDK, not Azure Bot Service/Bot Framework", "Can be complex to set up compared to simpler chatbot platforms", "Costs can escalate with Azure AI service usage", "Microsoft ecosystem lock-in and dependency", @@ -417,12 +431,13 @@ ], "metadata": { "license": "Proprietary (SDK is MIT open source)", + "status": "Legacy - Bot Framework SDK retired (final LTS ended 2025-12-31); superseded by Microsoft Copilot Studio and the Microsoft 365 Agents SDK", "supported_models": [ - "LUIS", "CLU", "Azure OpenAI", - "QnA Maker", - "Custom models" + "Custom models", + "LUIS (retired October 2025)", + "QnA Maker (retired March 2025)" ], "programming_languages": [ "C#", @@ -446,10 +461,14 @@ "Private endpoints" ], "development_tools": [ - "Bot Framework Composer", - "Bot Framework Emulator", + "Bot Framework Composer (legacy, no longer updated)", + "Bot Framework Emulator (legacy)", "VS Code extensions" ], + "successor_products": [ + "Microsoft Copilot Studio", + "Microsoft 365 Agents SDK" + ], "pricing": "Standard channels free, Premium channels charged per message. Bot Framework Composer included at no cost. Actual costs depend on Azure App Service and other Azure services used" }, "use_case_ratings": { @@ -495,13 +514,14 @@ } }, "best_for": [ - "Microsoft customers needing enterprise chatbot solutions", - "Teams building multi-channel conversational interfaces", + "Organizations maintaining existing Bot Framework bots while planning migration", + "Teams building multi-channel conversational interfaces (via Copilot Studio channel connectivity)", "Organizations requiring Azure ecosystem integration", - "Enterprises needing LUIS natural language understanding" + "New projects should evaluate Microsoft Copilot Studio or the Microsoft 365 Agents SDK instead" ], "tags": [ "microsoft", - "azure" + "azure", + "legacy" ] } diff --git a/data/agents/babyagi.json b/data/agents/babyagi.json index ff15db2..17d02b9 100644 --- a/data/agents/babyagi.json +++ b/data/agents/babyagi.json @@ -4,7 +4,7 @@ "name": "BabyAGI", "provider": "Yohei Nakajima", "version": "Classic", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "ARCHIVED: the original BabyAGI repo was archived to babyagi_archive in September 2024 and replaced by an experimental self-building framework; it is not production-maintained. Originally a minimalist autonomous task-driven AI agent that created, prioritized, and executed tasks toward an objective, demonstrating AGI concepts in under 200 lines of code.", "website": "https://github.com/yoheinakajima/babyagi", @@ -24,7 +24,7 @@ } ], "methodology": "Based on community testing and demonstrations", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 68, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 72, @@ -52,7 +52,7 @@ } ], "methodology": "Planning capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 70, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 55, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "task_generation": { "score": 75, @@ -94,7 +94,7 @@ } ], "methodology": "Task generation assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 60, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 55, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 65, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -169,7 +169,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 65, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 62, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 72, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 78, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 98, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "code_simplicity": { "score": 95, @@ -305,7 +305,7 @@ } ], "methodology": "Code complexity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 55, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 58, @@ -352,7 +352,7 @@ } ], "methodology": "Cost analysis", - "last_verified": "2025-11-09", + "last_verified": "2026-07-09", "notes": "Task generation can spiral, accumulating costs" }, "monitoring_capabilities": { @@ -367,7 +367,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 35, @@ -381,13 +381,13 @@ }, { "source": "GitHub Repository Status", - "url": "https://github.com/yoheinakajima/babyagi", - "date": "2026-06-10", - "value": "Original repo archived to babyagi_archive (Sept 2024); replaced by an experimental self-building framework; not production-maintained" + "url": "https://github.com/yoheinakajima/babyagi_archive", + "date": "2026-07-09", + "value": "Re-verified 2026-07-09: original repo remains archived in babyagi_archive (snapshot Sept 2024); current babyagi repo is an experimental self-building framework, not production-maintained" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Score reduced: original project archived and unmaintained since September 2024" } } diff --git a/data/agents/claude-agent-sdk.json b/data/agents/claude-agent-sdk.json index 0665a57..8f208f3 100644 --- a/data/agents/claude-agent-sdk.json +++ b/data/agents/claude-agent-sdk.json @@ -3,8 +3,8 @@ "type": "agent", "name": "Claude Agent SDK", "provider": "Anthropic", - "version": "0.x", - "last_evaluated": "2026-06-10", + "version": "0.3.x", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "SDK exposing Claude Code's production agent harness (tool loop, permission system, subagents, MCP) for building general-purpose agents in TypeScript and Python. Renamed from Claude Code SDK in September 2025.", "website": "https://code.claude.com/docs/en/agent-sdk/overview", @@ -24,7 +24,7 @@ } ], "methodology": "Evaluation of agents built on the SDK harness across coding and non-coding tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Testing of built-in tool loop, custom tool definitions, and MCP server integration", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "Long-horizon agent task evaluation using the SDK's managed loop and compaction", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Review of session resume, memory file support, and context compaction behavior", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 84, @@ -80,7 +80,7 @@ } ], "methodology": "Observed recovery from tool errors and failed commands in SDK-built agents", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 86, @@ -94,7 +94,7 @@ } ], "methodology": "Testing of programmatic subagent definition and parallel delegation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Review of inherited sandbox configuration and bash isolation options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 87, @@ -127,7 +127,7 @@ } ], "methodology": "Assessment of permission rules, programmatic approval callbacks, and hook-based gating", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 78, @@ -141,7 +141,7 @@ } ], "methodology": "Review of documented mitigations and developer responsibility for untrusted input handling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 80, @@ -155,7 +155,7 @@ } ], "methodology": "Architecture review of session/subagent context isolation in self-hosted deployments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 60, @@ -169,7 +169,7 @@ } ], "methodology": "License and source availability review of npm/PyPI packages", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Review of Anthropic API retention terms and local-state architecture", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 82, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance certification review including cloud-provider routing options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 78, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis across supported model endpoints", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 55, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment for self-hosted harness with cloud-only inference", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review across both language SDKs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 84, @@ -263,7 +263,7 @@ } ], "methodology": "Review of message stream observability and hook-based audit trails", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 82, @@ -277,7 +277,7 @@ } ], "methodology": "Assessment of streamed reasoning visibility and permission-request context", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 55, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment of SDK packages and underlying harness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 80, @@ -305,7 +305,7 @@ } ], "methodology": "Community engagement analysis via package downloads, release cadence, and ecosystem projects", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment from install to first working agent", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 84, @@ -338,7 +338,7 @@ } ], "methodology": "Assessment of horizontal scaling patterns and API rate limit constraints", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 70, @@ -349,10 +349,16 @@ "url": "https://www.anthropic.com/pricing", "date": "2026-05-01", "value": "Free SDK; costs are pay-as-you-go Claude tokens, which vary widely with agent autonomy and task length" + }, + { + "source": "Claude Agent SDK documentation", + "url": "https://code.claude.com/docs/en/agent-sdk/overview", + "date": "2026-07-09", + "value": "Since 2026-06-15, Agent SDK and non-interactive (claude -p) usage on subscription plans draws from a separate monthly Agent SDK credit ($20 on Pro, $100 on Max 5x, $200 on Max 20x), capping subscription-based SDK spend" } ], - "methodology": "Pricing model analysis of token-based costs for long-running agents", - "last_verified": "2026-06-10" + "methodology": "Pricing model analysis of token-based costs for long-running agents, including the June 2026 subscription-credit change", + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 78, @@ -366,7 +372,7 @@ } ], "methodology": "Review of built-in usage reporting and integration points for external observability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 85, @@ -380,7 +386,7 @@ } ], "methodology": "Maturity assessment based on shared production harness and release stability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -444,7 +450,7 @@ "Subagents" ], "first_release": "2025 (as Claude Code SDK); renamed Claude Agent SDK 2025-09-29", - "pricing": "Free SDK; pay-as-you-go Claude token costs", + "pricing": "Free SDK; pay-as-you-go Claude token costs via API, or monthly Agent SDK credits on subscription plans since 2026-06-15 ($20 Pro / $100 Max 5x / $200 Max 20x)", "packages": [ "@anthropic-ai/claude-agent-sdk (npm)", "claude-agent-sdk (PyPI)" diff --git a/data/agents/claude-code.json b/data/agents/claude-code.json index a14a786..7ef60ae 100644 --- a/data/agents/claude-code.json +++ b/data/agents/claude-code.json @@ -3,8 +3,8 @@ "type": "agent", "name": "Claude Code", "provider": "Anthropic", - "version": "2.x", - "last_evaluated": "2026-06-10", + "version": "2.1.x (2.1.204 as of 2026-07-08)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Anthropic's agentic coding tool available as a terminal CLI, IDE extensions, web, and desktop app. Plans and executes multi-step coding tasks with tiered permissions, OS-level sandboxing, MCP integration, hooks, subagents, and plugins/skills.", "website": "https://www.anthropic.com/claude-code", @@ -30,7 +30,7 @@ } ], "methodology": "Benchmark results review plus adoption and revenue signals as proxy for sustained task success in production use", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 90, @@ -44,7 +44,7 @@ } ], "methodology": "Hands-on testing of built-in tools and MCP integrations across coding workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 90, @@ -58,7 +58,7 @@ } ], "methodology": "Evaluation of plan mode and long-horizon task execution on multi-file repository changes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 82, @@ -72,7 +72,7 @@ } ], "methodology": "Review of memory file hierarchy, context compaction behavior, and cross-session resume", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 85, @@ -86,7 +86,7 @@ } ], "methodology": "Observed recovery behavior from failing builds, tests, and tool errors during evaluation sessions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 87, @@ -100,15 +100,15 @@ } ], "methodology": "Testing of subagent delegation, parallel task fan-out, and plugin-defined agents", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 78, + "overall_score": 77, "criteria": { "tool_sandboxing": { - "score": 88, + "score": 84, "confidence": "high", "evidence": [ { @@ -122,10 +122,16 @@ "url": "https://www.anthropic.com/news/claude-code-on-the-web", "date": "2025-10-20", "value": "Web version executes tasks inside Anthropic-managed isolated sandboxes" + }, + { + "source": "SentinelOne vulnerability database - CVE-2026-39861", + "url": "https://www.sentinelone.com/vulnerability-database/cve-2026-39861/", + "date": "2026-07-09", + "value": "Two CVSS 10.0 sandbox-escape CVEs disclosed and patched in 2026 (CVE-2026-39861 symlink escape; CVE-2026-25725 settings.json protection bypass); ~28 CVEs total in the product's first year, with rapid vendor patching" } ], - "methodology": "Review of sandbox architecture (filesystem and network isolation) and managed cloud sandbox design", - "last_verified": "2026-06-10" + "methodology": "Review of sandbox architecture (filesystem and network isolation), managed cloud sandbox design, and 2026 CVE history; score reduced modestly (88 to 84) to reflect two max-severity sandbox escapes, balanced by fast patch turnaround", + "last_verified": "2026-07-09" }, "access_control": { "score": 85, @@ -139,7 +145,7 @@ } ], "methodology": "Assessment of permission model, allowlist granularity, and enterprise policy controls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 80, @@ -153,7 +159,7 @@ } ], "methodology": "Review of documented mitigations and behavior when processing untrusted repository and web content", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 82, @@ -167,7 +173,7 @@ } ], "methodology": "Architecture review of session isolation in cloud sandboxes and local filesystem scoping", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 55, @@ -181,7 +187,7 @@ } ], "methodology": "License and source availability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -200,7 +206,7 @@ } ], "methodology": "Review of Anthropic data retention commitments across consumer and commercial tiers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 82, @@ -214,7 +220,7 @@ } ], "methodology": "Compliance certification and DPA availability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 80, @@ -228,7 +234,7 @@ } ], "methodology": "Data flow analysis of code, prompt, and telemetry handling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 45, @@ -242,7 +248,7 @@ } ], "methodology": "Deployment options assessment including Bedrock/Vertex routing and air-gap feasibility", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -261,7 +267,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 85, @@ -275,7 +281,7 @@ } ], "methodology": "Review of session transcripts, hooks-based auditing, and OTel telemetry support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 84, @@ -289,7 +295,7 @@ } ], "methodology": "Assessment of plan previews, inline reasoning, and diff-based change explanation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 50, @@ -303,21 +309,21 @@ } ], "methodology": "Open source assessment of core product and ecosystem components", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 90, "confidence": "high", "evidence": [ { - "source": "Claude Code GitHub repository", - "url": "https://github.com/anthropics/claude-code", - "date": "2026-06-01", - "value": "Highly active issue tracker, frequent releases, and a large plugin/skills ecosystem since launch" + "source": "Claude Code changelog", + "url": "https://code.claude.com/docs/en/changelog", + "date": "2026-07-08", + "value": "v2.1.204 released 2026-07-08; sustained near-daily release cadence, highly active issue tracker, and a large plugin/skills ecosystem" } ], "methodology": "Community engagement analysis via GitHub activity, release cadence, and ecosystem growth", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -336,7 +342,7 @@ } ], "methodology": "Setup time and integration surface assessment across CLI, IDE, web, and desktop", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 85, @@ -350,7 +356,7 @@ } ], "methodology": "Assessment of parallel cloud sessions, headless/CI usage, and rate limit behavior", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 72, @@ -364,7 +370,7 @@ } ], "methodology": "Pricing model analysis comparing subscription caps versus variable API token costs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 80, @@ -378,7 +384,7 @@ } ], "methodology": "Review of OTel export, cost/usage tracking, and admin analytics features", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 90, @@ -392,7 +398,7 @@ } ], "methodology": "Maturity assessment from GA timeline, release stability, and enterprise adoption", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -434,7 +440,8 @@ "Locked to Claude models; no local or third-party model support", "API pay-as-you-go costs can spike on large autonomous tasks", "Subscription rate limits can interrupt heavy daily usage", - "Autonomous edits still require human review for correctness and security" + "Autonomous edits still require human review for correctness and security", + "Notable 2026 CVE history (~28 CVEs in first year, including two patched CVSS 10.0 sandbox escapes: CVE-2026-39861, CVE-2026-25725) and an accidental full source-map leak in npm v2.1.88 (2026-03-31); patches shipped quickly but the attack surface is large" ], "metadata": { "license": "Proprietary (public releases repo at github.com/anthropics/claude-code)", diff --git a/data/agents/crewai.json b/data/agents/crewai.json index c77848e..62e14a6 100644 --- a/data/agents/crewai.json +++ b/data/agents/crewai.json @@ -3,10 +3,10 @@ "type": "agent", "name": "CrewAI", "provider": "CrewAI Inc.", - "version": "1.2.1", - "last_evaluated": "2025-11-09", + "version": "1.14.6", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Role-playing multi-agent framework for orchestrating collaborative autonomous agents. Agents work together as a crew with defined roles, goals, and backstories to tackle complex tasks through delegation and collaboration.", + "description": "Role-playing multi-agent framework for orchestrating collaborative autonomous agents. Agents work as a crew with defined roles, goals, and backstories to tackle complex tasks through delegation. Commercial offerings: the CrewAI AMP managed platform and self-hosted CrewAI Factory. Note: four Code Interpreter CVEs (CVE-2026-2275/2285/2286/2287) disclosed 2026-03-30; vendor reports all fixed in current releases, with the built-in CodeInterpreterTool removed in favor of external sandboxes.", "website": "https://www.crewai.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on multi-agent coordination testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 82, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 78, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 76, @@ -74,13 +74,13 @@ "evidence": [ { "source": "Community Reports", - "url": "https://github.com/joaomdmoura/crewAI/issues", + "url": "https://github.com/crewAIInc/crewAI/issues", "date": "2024-09-20", "value": "Basic error handling, agent delegation can help recovery" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 88, @@ -94,26 +94,39 @@ } ], "methodology": "Multi-agent coordination testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 74, + "overall_score": 72, "criteria": { "tool_sandboxing": { - "score": 68, - "confidence": "medium", + "score": 60, + "confidence": "high", "evidence": [ { "source": "CrewAI Architecture", "url": "https://docs.crewai.com/", "date": "2024-10-01", "value": "No built-in sandboxing, relies on tool implementation" + }, + { + "source": "CERT/CC VU#221883", + "url": "https://www.kb.cert.org/vuls/id/221883", + "date": "2026-03-30", + "value": "CVE-2026-2275, CVE-2026-2285, CVE-2026-2286, CVE-2026-2287: Code Interpreter fell back to unsandboxed execution when Docker unavailable (RCE), arbitrary local file read in JSON loader, and SSRF in RAG search tools via prompt injection" + }, + { + "source": "CERT/CC VU#221883 (vendor statement)", + "url": "https://www.kb.cert.org/vuls/id/221883", + "date": "2026-05-20", + "value": "Vendor reports all issues fixed in current releases: CodeInterpreterTool removed entirely, allow_code_execution deprecated, centralized path/URL validation added; external sandboxes (E2B, Daytona) recommended for code execution" } ], - "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "methodology": "Security architecture review plus disclosed vulnerability analysis", + "last_verified": "2026-07-09", + "notes": "Score lowered 68 to 60: March 2026 CVE cluster showed sandbox fell back open (unsandboxed execution) under Docker failure; remediated by removing the built-in code interpreter, which shifts sandboxing responsibility to external providers" }, "access_control": { "score": 72, @@ -127,7 +140,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 75, @@ -141,7 +154,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 78, @@ -155,7 +168,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -163,13 +176,13 @@ "evidence": [ { "source": "CrewAI GitHub", - "url": "https://github.com/joaomdmoura/crewAI", + "url": "https://github.com/crewAIInc/crewAI", "date": "2024-10-20", "value": "Open source MIT license, 20k+ stars, active community" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +201,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 82, @@ -196,13 +209,13 @@ "evidence": [ { "source": "Open Source Framework", - "url": "https://github.com/joaomdmoura/crewAI", + "url": "https://github.com/crewAIInc/crewAI", "date": "2024-10-01", "value": "GDPR compliance possible with proper configuration" } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 73, @@ -216,7 +229,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 90, @@ -230,7 +243,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +262,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 78, @@ -263,7 +276,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +290,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 92, @@ -285,13 +298,13 @@ "evidence": [ { "source": "GitHub Repository", - "url": "https://github.com/joaomdmoura/crewAI", + "url": "https://github.com/crewAIInc/crewAI", "date": "2024-10-20", "value": "MIT licensed, 20k+ stars, very active development" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_activity": { "score": 88, @@ -299,13 +312,13 @@ "evidence": [ { "source": "GitHub Activity", - "url": "https://github.com/joaomdmoura/crewAI", + "url": "https://github.com/crewAIInc/crewAI", "date": "2024-10-20", "value": "Very active community with frequent updates" } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +337,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 75, @@ -332,13 +345,13 @@ "evidence": [ { "source": "Community Discussions", - "url": "https://github.com/joaomdmoura/crewAI/discussions", + "url": "https://github.com/crewAIInc/crewAI/discussions", "date": "2024-09-15", "value": "Scalability depends on infrastructure and LLM rate limits" } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -346,13 +359,13 @@ "evidence": [ { "source": "Open Source Pricing", - "url": "https://github.com/joaomdmoura/crewAI", + "url": "https://github.com/crewAIInc/crewAI", "date": "2024-10-01", "value": "Free framework, costs only from LLM API calls" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 72, @@ -366,7 +379,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 76, @@ -374,13 +387,19 @@ "evidence": [ { "source": "Framework Maturity", - "url": "https://github.com/joaomdmoura/crewAI", + "url": "https://github.com/crewAIInc/crewAI", "date": "2024-10-20", "value": "Rapidly evolving framework, some API changes between versions" + }, + { + "source": "crewai on PyPI", + "url": "https://pypi.org/project/crewai/", + "date": "2026-07-09", + "value": "Stable 1.x series: latest release 1.14.6 (2026-05-28); repo very active (~55,200 stars, pushed 2026-07-09); enterprise AMP platform and self-hosted Factory available" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -442,6 +461,7 @@ "Good integration with LangChain tools ecosystem" ], "limitations": [ + "Four CVEs disclosed 2026-03-30 (RCE via Code Interpreter sandbox fallback, arbitrary file read, SSRF); vendor fixed by removing the built-in CodeInterpreterTool, so code execution now requires an external sandbox (E2B, Daytona)", "Can be expensive with multiple agents making LLM calls", "Agent coordination overhead can increase latency", "Rapidly evolving API may require code updates", @@ -466,11 +486,16 @@ "Custom tools", "Built-in tools" ], - "github_stars": "39900+", + "github_stars": "55200+", "first_release": "2023", - "pricing": "Free (MIT license) - Costs only from LLM API calls", + "current_version": "1.14.6 (2026-05-28)", + "pricing": "Framework free (MIT) - Costs from LLM API calls. Managed CrewAI AMP platform: Free tier (50 executions/month), paid tiers from ~$25-99/month, custom Enterprise pricing; self-hosted CrewAI Factory for on-prem/private cloud", + "pricing_last_verified": "2026-07-09", + "funding": "$18M total including Series A led by Insight Partners (announced 2024-10-22)", + "enterprise_offering": "CrewAI AMP (managed SaaS) and CrewAI Factory (containerized self-hosted) with visual editor, observability, triggers, and guardrails", "python_requirement": "Python >=3.10 <3.14", - "adoption": "Powers 1.4B+ agentic automations globally" + "adoption": "Powers 1.4B+ agentic automations globally (vendor claim)", + "security_advisories": "CVE-2026-2275, CVE-2026-2285, CVE-2026-2286, CVE-2026-2287 (CERT VU#221883, disclosed 2026-03-30; vendor reports fixed in current releases as of 2026-05-20)" }, "tags": [ "multi-agent", diff --git a/data/agents/cursor-agent.json b/data/agents/cursor-agent.json index acc643e..039b795 100644 --- a/data/agents/cursor-agent.json +++ b/data/agents/cursor-agent.json @@ -2,11 +2,11 @@ "id": "cursor-agent", "type": "agent", "name": "Cursor Agent", - "provider": "Anysphere", + "provider": "Anysphere (SpaceX acquisition announced 2026-06-16, expected to close Q3 2026)", "version": "3.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Agent mode of Cursor, Anysphere's AI-native IDE. The product's primary surface is now agentic: parallel local agents, cloud/background agents running in isolated VMs, and the in-house Composer model line alongside Claude, GPT, and Gemini.", + "description": "Agent mode of Cursor, Anysphere's AI-native IDE. The product's primary surface is now agentic: parallel local agents, cloud/background agents running in isolated VMs, and the in-house Composer model line (Composer 2.5) alongside Claude, GPT, and Gemini. Anysphere IPO'd on Nasdaq in June 2026 and days later agreed to a $60B all-stock acquisition by SpaceX (announced 2026-06-16, expected to close Q3 2026 under its xAI subsidiary).", "website": "https://cursor.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Assessment of agentic edit and task completion quality across Composer and frontier model options, drawing on vendor benchmarks and user reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "Review of agent tool loop reliability across edit, search, terminal, and MCP integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Evaluation of plan construction and multi-file refactoring on complex tasks in the agent-first interface", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 78, @@ -66,7 +66,7 @@ } ], "methodology": "Review of rules, memories, and codebase indexing persistence", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 81, @@ -80,7 +80,7 @@ } ], "methodology": "Assessment of automatic iteration on lints, build errors, and failing tests", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 82, @@ -94,7 +94,7 @@ } ], "methodology": "Review of parallel and background agent orchestration capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security review of cloud VM isolation versus local execution approval model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 76, @@ -127,7 +127,7 @@ } ], "methodology": "Review of org-level controls, SSO, and repository scoping", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 68, @@ -141,7 +141,7 @@ } ], "methodology": "Threat surface analysis of untrusted content ingestion against documented mitigations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 74, @@ -155,7 +155,7 @@ } ], "methodology": "Review of privacy mode guarantees and tenant isolation for cloud agents", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 32, @@ -169,7 +169,7 @@ } ], "methodology": "Source availability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Review of privacy mode retention guarantees and default settings", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 78, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance documentation assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 66, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis across multi-provider model routing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 35, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 80, @@ -263,7 +263,7 @@ } ], "methodology": "Review of action visibility, diff review workflow, and agent logs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 76, @@ -277,7 +277,7 @@ } ], "methodology": "Assessment of plan narration and change rationale quality", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 25, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 80, @@ -301,11 +301,11 @@ "source": "Cursor Community Forum", "url": "https://forum.cursor.com/", "date": "2026-06-01", - "value": "Very large active user community, busy forum, and rapid release cadence (Cursor 2.0 Oct 2025, Composer 2 Mar 2026, Cursor 3 Apr 2026)" + "value": "Very large active user community, busy forum, and rapid release cadence (Cursor 2.0 Oct 2025, Composer 2 Mar 2026, Cursor 3 Apr 2026, Composer 2.5 mid-2026)" } ], "methodology": "Community engagement and release cadence analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Onboarding and integration friction assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 80, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability assessment of parallel and cloud agent execution", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 70, @@ -349,10 +349,16 @@ "url": "https://cursor.com/pricing", "date": "2026-04-15", "value": "Free / Pro $20 / Pro+ $60 / Ultra $200 per month; Teams $40 per seat. Tiers include usage allowances with overage billing for heavy frontier-model use" + }, + { + "source": "Cursor Blog - Improvements to Teams Pricing", + "url": "https://cursor.com/blog/teams-pricing-june-2026", + "date": "2026-06-30", + "value": "June 2026 Teams overhaul (effective 2026-07-01 for renewals): Standard seat $40/mo ($32 annual) with split usage pools for Composer/Auto vs third-party API models; new Premium seat at $120/mo ($96 annual) with 5x usage; Cursor estimates lower costs for ~90% of teams" } ], - "methodology": "Pricing model analysis; flat tiers are clear but usage-based overages reduce predictability for heavy agent users", - "last_verified": "2026-06-10" + "methodology": "Pricing model analysis including the June 2026 Teams restructure; flat tiers are clear but usage-based overages reduce predictability for heavy agent users", + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 72, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring and admin analytics features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 85, @@ -377,10 +383,16 @@ "url": "https://www.infoq.com/news/2026/04/cursor-3-agent-first-interface/", "date": "2026-04-20", "value": "Mature, widely adopted product with massive enterprise and individual user base and sustained rapid iteration through Cursor 3" + }, + { + "source": "TechCrunch - SpaceX to acquire Cursor for $60B", + "url": "https://techcrunch.com/2026/06/16/spacex-to-acquire-cursor-for-60b-in-stock-days-after-blockbuster-ipo/", + "date": "2026-06-16", + "value": "Annualized revenue reached ~$4B by early June 2026; Nasdaq IPO June 2026 followed by $60B all-stock SpaceX acquisition agreement (close expected Q3 2026), introducing ownership-transition uncertainty alongside strong commercial traction" } ], - "methodology": "Product maturity and adoption assessment", - "last_verified": "2026-06-10" + "methodology": "Product maturity and adoption assessment; score held at 85 as strong adoption and revenue are offset by pending ownership change", + "last_verified": "2026-07-09" } } } @@ -411,7 +423,7 @@ ], "strengths": [ "Agent-first IDE redesign (Cursor 3, April 2026) makes agent orchestration the primary workflow", - "In-house Composer model line (Composer 2, 2026-03-19) delivers very fast agentic edits alongside Claude/GPT/Gemini choice", + "In-house Composer model line (latest Composer 2.5) delivers very fast agentic edits alongside Claude/GPT/Gemini choice", "Parallel agents and cloud background agents in isolated VMs scale work beyond one task at a time", "Human-in-the-loop by design: reviewable diffs, command approvals, and visible terminal output", "Seamless VS Code compatibility for extensions, themes, and keybindings", @@ -423,12 +435,13 @@ "Usage-based overages above tier allowances make heavy agent usage costs less predictable", "Local agent terminal execution depends on user-configured approvals; misconfiguration widens risk", "Multi-provider model routing complicates data governance reviews", - "Rapid release cadence occasionally introduces regressions and workflow changes" + "Rapid release cadence occasionally introduces regressions and workflow changes", + "Pending SpaceX/xAI acquisition (announced 2026-06-16, closing Q3 2026) creates governance, roadmap, and data-stewardship uncertainty for enterprise buyers" ], "metadata": { "license": "Proprietary", "supported_models": [ - "Composer 1/2 (Anysphere in-house)", + "Composer 1/2/2.5 (Anysphere in-house)", "Anthropic Claude", "OpenAI GPT", "Google Gemini" @@ -445,8 +458,8 @@ "GitHub integration for cloud agents" ], "first_release": "2023 (IDE); Cursor 2.0 with Composer Oct 2025; Cursor 3 April 2026", - "pricing": "Free / Pro $20 / Pro+ $60 / Ultra $200 per month; Teams $40 per seat", - "company": "Anysphere" + "pricing": "Free / Pro $20 / Pro+ $60 / Ultra $200 per month; Teams (June 2026): Standard $40/mo ($32 annual), Premium $120/mo ($96 annual) per seat", + "company": "Anysphere (Nasdaq IPO June 2026; ~$4B ARR; $60B all-stock SpaceX acquisition announced 2026-06-16, expected close Q3 2026 under xAI subsidiary)" }, "related_entities": [ "claude-code", diff --git a/data/agents/devin.json b/data/agents/devin.json index c11ac81..7feed67 100644 --- a/data/agents/devin.json +++ b/data/agents/devin.json @@ -4,9 +4,9 @@ "name": "Devin", "provider": "Cognition", "version": "2.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Autonomous AI software engineer from Cognition that plans and executes multi-step engineering tasks in a sandboxed cloud workspace with its own editor, shell, and browser, and delivers work as pull requests.", + "description": "Autonomous AI software engineer from Cognition that plans and executes multi-step engineering tasks in a sandboxed cloud workspace with its own editor, shell, and browser, and delivers work as pull requests. Now spans Devin Cloud agents, Devin Desktop (the rebranded Windsurf IDE, June 2026) with the Rust-based Devin Local agent, and Cognition's in-house SWE model line (SWE-1.7, July 2026).", "website": "https://devin.ai/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Assessment of task completion on scoped engineering work based on vendor documentation, customer case studies, and independent user reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 84, @@ -38,7 +38,7 @@ } ], "methodology": "Review of integrated toolchain reliability across shell, browser, and VCS operations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 86, @@ -52,7 +52,7 @@ } ], "methodology": "Evaluation of plan generation, user-editable plans, and plan adherence on long-horizon tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Review of cross-session memory features (Knowledge, Playbooks, Wiki) and session snapshot persistence", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 76, @@ -80,7 +80,7 @@ } ], "methodology": "Assessment of autonomous debugging behavior and failure-mode reports from production users", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 74, @@ -94,7 +94,7 @@ } ], "methodology": "Review of parallel session capabilities and multi-Devin task delegation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of isolated cloud workspace model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 80, @@ -127,7 +127,7 @@ } ], "methodology": "Review of identity, repository scoping, and secrets handling controls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 70, @@ -141,7 +141,7 @@ } ], "methodology": "Threat surface analysis of autonomous browsing and untrusted repo content; limited public disclosure of defenses", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 78, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review of tenant and session isolation claims", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 30, @@ -165,11 +165,11 @@ "source": "Cognition", "url": "https://cognition.ai/", "date": "2026-05-27", - "value": "Fully proprietary product and models (in-house SWE-1.5 line plus frontier models); no source code or model weights published" + "value": "Fully proprietary product and models (in-house SWE model line, latest SWE-1.7, plus frontier models); no source code or model weights published" } ], "methodology": "Source availability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Review of published retention practices and enterprise data controls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 72, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance documentation assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 65, @@ -212,11 +212,11 @@ "source": "Cognition - Devin Product Page", "url": "https://devin.ai/", "date": "2026-05-01", - "value": "Code is processed by Cognition's own SWE-1.5 model line and routed to third-party frontier models for some tasks" + "value": "Code is processed by Cognition's own SWE model line (latest SWE-1.7) and routed to third-party models (OpenAI, Claude, Gemini options on paid plans) for some tasks" } ], "methodology": "Data flow analysis of model routing between in-house and third-party providers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 40, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 85, @@ -263,7 +263,7 @@ } ], "methodology": "Review of session visibility, live workspace observation, and replay features", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 82, @@ -277,7 +277,7 @@ } ], "methodology": "Assessment of plan transparency and change justification quality", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 20, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 65, @@ -305,7 +305,7 @@ } ], "methodology": "Community and ecosystem engagement analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration surface assessment across team workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 82, @@ -338,21 +338,21 @@ } ], "methodology": "Scalability assessment of parallel cloud session model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { - "score": 62, + "score": 66, "confidence": "high", "evidence": [ { "source": "Devin Pricing", "url": "https://devin.ai/pricing", - "date": "2026-05-01", - "value": "Core plan from $20 pay-as-you-go at $2.25/ACU (Devin 2.0, April 2025, down from $500/mo); Team $500/mo. ACU consumption varies widely by task complexity" + "date": "2026-07-09", + "value": "Restructured plans: Free $0; Pro $20/mo (full model access, Devin Cloud agents); Max $200/mo (higher quotas); Teams $80/mo base + $40/mo per full developer seat; Enterprise custom. Quota-based usage refreshing daily/weekly replaces headline ACU pricing (ACUs remain the underlying meter, ~$2.00-2.25/ACU on team tiers)" } ], - "methodology": "Pricing model analysis; ACU-metered billing makes per-task costs hard to forecast", - "last_verified": "2026-06-10" + "methodology": "Pricing model analysis; 2026 quota-based subscription tiers improve predictability over pure ACU metering (score raised 62 to 66), though heavy usage still burns variable quota per task", + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 78, @@ -366,7 +366,7 @@ } ], "methodology": "Monitoring and usage governance features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 80, @@ -380,7 +380,7 @@ } ], "methodology": "Vendor maturity and product stability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -413,12 +413,12 @@ "True end-to-end autonomy: plans, codes, tests, browses docs, and opens PRs in its own cloud workspace", "Sandboxed cloud VMs isolate execution from user infrastructure", "Interactive, editable plans and fully replayable session timelines provide strong traceability", - "Devin 2.0 pricing ($20 entry, $2.25/ACU) dramatically lowered the adoption barrier from the original $500/mo", + "Accessible pricing: Free tier and $20/mo Pro plan (vs the original $500/mo), with quota-based subscriptions since mid-2026", "Persistent Knowledge, Playbooks, and auto-generated Devin Wiki retain organizational context", "Strong vendor trajectory: $26B valuation, ~$492M ARR, Windsurf acquisition (2025-07-14)" ], "limitations": [ - "ACU-metered billing makes costs unpredictable, especially when the agent pursues unproductive paths", + "Usage-quota/ACU-metered consumption makes costs unpredictable, especially when the agent pursues unproductive paths", "Fully proprietary stack with no self-hosted option; code must be processed in Cognition's cloud", "Reliability drops on ambiguous or large unscoped tasks, requiring careful task decomposition", "Prompt injection defenses for autonomous browsing are not publicly documented", @@ -428,8 +428,8 @@ "metadata": { "license": "Proprietary", "supported_models": [ - "Cognition SWE-1.5 model line (in-house)", - "Third-party frontier models for select tasks" + "Cognition SWE model line (in-house; SWE-1.7 launched 2026-07-08, served at ~1,000 tokens/sec via Cerebras)", + "OpenAI, Anthropic Claude, and Google Gemini models on paid plans" ], "programming_languages": [ "Most major languages (Python, TypeScript, Java, Go, etc.)" @@ -443,8 +443,8 @@ "API access" ], "first_release": "2024 (limited), Devin 2.0 April 2025", - "pricing": "Core from $20 pay-as-you-go ($2.25/ACU); Team $500/mo; Enterprise custom", - "company_milestones": "Acquired Windsurf 2025-07-14; raised $1B+ at $26B valuation (closed 2026-05-27); ~$492M ARR" + "pricing": "Free $0; Pro $20/mo; Max $200/mo; Teams $80/mo base + $40/mo per full developer seat; Enterprise custom (quota-based, ACU-metered underneath)", + "company_milestones": "Acquired Windsurf 2025-07-14 (rebranded Devin Desktop 2026-06-02, with Rust-based Devin Local agent); raised $1B+ at $26B valuation (closed 2026-05-27); ~$492M ARR; AI Productivity Guarantee (up to $10M in usage credits) announced June 2026; SWE-1.7 model launched 2026-07-08" }, "related_entities": [ "openai-codex", diff --git a/data/agents/dify.json b/data/agents/dify.json index 6c02146..cb823a7 100644 --- a/data/agents/dify.json +++ b/data/agents/dify.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Dify", "provider": "LangGenius", - "version": "1.x", - "last_evaluated": "2026-06-10", + "version": "1.15.0", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source LLM application and agentic workflow platform with a visual canvas, built-in RAG pipeline, agent nodes, and 50+ built-in tools. One of the most-starred LLM app platforms, deployable self-hosted or via Dify Cloud.", + "description": "Open-source LLM application and agentic workflow platform with a visual canvas, built-in RAG pipeline, agent nodes, and 50+ tools. One of the most-starred LLM app platforms (148k+ stars), self-hosted or via Dify Cloud. Moderate-severity 2026 advisories (authorization bypass CVE-2026-41949, path traversal CVE-2026-41948, XSS CVE-2026-6619, account enumeration CVE-2026-28288) make staying on current releases (1.15.0+) important; no critical RCE or in-the-wild exploitation reported.", "website": "https://dify.ai/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Task completion testing across chatflow, workflow, and agent app types", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "Tool invocation testing across built-in, custom, and plugin tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Complex workflow construction and execution testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Memory evaluation across sessions and knowledge bases", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 78, @@ -80,7 +80,7 @@ } ], "methodology": "Error injection testing on workflow error branches and retries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 72, @@ -94,7 +94,7 @@ } ], "methodology": "Multi-agent pattern testing using agent nodes within workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of code execution sandbox and tool isolation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 78, @@ -124,10 +124,17 @@ "url": "https://docs.dify.ai/", "date": "2026-04-20", "value": "Workspace member roles, app-level API keys, and SSO/access policies in premium/enterprise editions" + }, + { + "source": "Dify 2026 security advisories", + "url": "https://github.com/langgenius/dify/security", + "date": "2026-07-09", + "value": "Moderate 2026 advisories: CVE-2026-41949 (authorization bypass in file preview endpoint, <=1.14.1), CVE-2026-41948 (authenticated path traversal to Plugin Daemon API, <=1.14.1), CVE-2026-28288 (account enumeration, fixed in 1.9.0), CVE-2026-6619 (XSS in ImagePreview, <=1.13.3)" } ], "methodology": "Access control assessment of workspace roles and API key scoping", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09", + "notes": "Score held at 78: 2026 advisories are moderate severity (no critical RCE, no known in-the-wild exploitation) and addressed in current releases; keep deployments on 1.15.0+" }, "prompt_injection_defense": { "score": 70, @@ -141,7 +148,7 @@ } ], "methodology": "Injection testing with moderation toolkits enabled", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 76, @@ -155,7 +162,7 @@ } ], "methodology": "Data isolation architecture review across cloud and self-hosted modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 84, @@ -164,12 +171,12 @@ { "source": "Dify GitHub Repository", "url": "https://github.com/langgenius/dify", - "date": "2026-06-01", - "value": "Fully public codebase with 138k+ stars; license is Apache-2.0-based but adds multi-tenant SaaS resale restrictions" + "date": "2026-07-09", + "value": "Fully public codebase with 148k+ stars; license is Apache-2.0-based but adds multi-tenant SaaS resale restrictions" } ], "methodology": "Source code and license terms review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +195,7 @@ } ], "methodology": "Privacy architecture review of self-hosted versus cloud retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 81, @@ -202,7 +209,7 @@ } ], "methodology": "Compliance capabilities assessment across deployment modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 74, @@ -216,7 +223,7 @@ } ], "methodology": "Data flow analysis across model providers and tool integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 92, @@ -230,7 +237,7 @@ } ], "methodology": "Deployment options assessment including air-gapped configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +256,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 84, @@ -263,7 +270,7 @@ } ], "methodology": "Tracing and logging capabilities assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +284,7 @@ } ], "methodology": "Explainability assessment of visual runs and RAG citations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 82, @@ -291,7 +298,7 @@ } ], "methodology": "License terms review against OSI-standard licenses", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 92, @@ -300,12 +307,12 @@ { "source": "Dify GitHub Metrics", "url": "https://github.com/langgenius/dify", - "date": "2026-04-15", - "value": "138k+ GitHub stars and 1M+ deployed apps as of April 2026; very active releases and plugin marketplace" + "date": "2026-07-09", + "value": "148k+ GitHub stars (July 2026) and 1M+ deployed apps as of April 2026; very active releases (1.15.0 shipped June 25, 2026) and plugin marketplace" } ], "methodology": "Community engagement analysis of stars, deployments, and release cadence", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +331,7 @@ } ], "methodology": "Integration complexity assessment for no-code and API usage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 82, @@ -338,7 +345,7 @@ } ], "methodology": "Scalability assessment of deployment architectures", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 84, @@ -352,7 +359,7 @@ } ], "methodology": "Pricing model analysis across self-hosted and cloud tiers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 84, @@ -366,7 +373,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 84, @@ -380,7 +387,7 @@ } ], "methodology": "Production readiness assessment of release maturity and adoption", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -423,7 +430,7 @@ ], "strengths": [ "Visual workflow canvas combining agent nodes, RAG, and 50+ built-in tools", - "Massive adoption and community: 138k+ GitHub stars, 1M+ deployed apps", + "Massive adoption and community: 148k+ GitHub stars, 1M+ deployed apps", "Free self-hosting with Docker/Helm and full data control", "Built-in observability: logs, annotations, analytics, and tracing integrations", "Dedicated dify-sandbox for code-node execution", @@ -434,7 +441,8 @@ "Multi-agent orchestration is shallower than code-first frameworks like LangGraph or CrewAI", "Visual abstraction limits fine-grained programmatic control for complex agent logic", "Prompt injection defense relies on opt-in moderation rather than dedicated mechanisms", - "Self-hosted stack (API, worker, sandbox, vector DB) has nontrivial operational footprint" + "Self-hosted stack (API, worker, sandbox, vector DB) has nontrivial operational footprint", + "Steady stream of moderate-severity security advisories in 2026 (authorization bypass, path traversal, XSS, account enumeration) requires keeping deployments on current releases (1.15.0+)" ], "related": [ "flowise", @@ -465,8 +473,9 @@ "Plugin marketplace", "MCP tools" ], - "github_stars": "138000+", + "github_stars": "148000+", "first_release": "2023", + "latest_version": "1.15.0 (June 25, 2026)", "pricing": "Free self-hosted; Cloud: Sandbox free, Professional $59/mo, Team $159/mo", "python_requirement": "Python >=3.11 (backend, when developing from source)", "adoption": "1M+ apps deployed on the platform as of April 2026" diff --git a/data/agents/e2b-agents.json b/data/agents/e2b-agents.json index 87f2cef..20b459c 100644 --- a/data/agents/e2b-agents.json +++ b/data/agents/e2b-agents.json @@ -4,7 +4,7 @@ "name": "E2B Agents", "provider": "E2B", "version": "Cloud SDK", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Secure cloud runtime for AI agents with code interpreter capabilities. Provides sandboxed environments for executing agent-generated code safely, with support for multiple programming languages and pre-built integrations.", "website": "https://e2b.dev/", @@ -24,7 +24,7 @@ } ], "methodology": "Code execution testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sandbox_isolation": { "score": 92, @@ -38,7 +38,7 @@ } ], "methodology": "Isolation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_language_support": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Language capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "startup_latency": { "score": 76, @@ -66,7 +66,7 @@ } ], "methodology": "Latency testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "file_system_support": { "score": 84, @@ -80,7 +80,7 @@ } ], "methodology": "File operations testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "50-500ms (warm), 1-3s (cold start)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "network_isolation": { "score": 90, @@ -127,7 +127,7 @@ } ], "methodology": "Network security testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "code_sandboxing": { "score": 94, @@ -141,7 +141,7 @@ } ], "methodology": "Sandbox security testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_authentication": { "score": 88, @@ -155,7 +155,7 @@ } ], "methodology": "Authentication testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_encryption": { "score": 86, @@ -169,7 +169,7 @@ } ], "methodology": "Encryption assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 79, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ephemeral_environments": { "score": 90, @@ -216,7 +216,7 @@ } ], "methodology": "Data lifecycle assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "code_privacy": { "score": 76, @@ -230,7 +230,7 @@ } ], "methodology": "Code privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 75, @@ -244,7 +244,7 @@ } ], "methodology": "Logging assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sdk_support": { "score": 86, @@ -277,7 +277,7 @@ } ], "methodology": "SDK assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_sdk": { "score": 84, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_visibility": { "score": 80, @@ -305,7 +305,7 @@ } ], "methodology": "Traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_examples": { "score": 78, @@ -319,7 +319,7 @@ } ], "methodology": "Community resources assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 85, @@ -352,21 +352,21 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 78, "confidence": "medium", "evidence": [ { - "source": "Pricing", + "source": "E2B Pricing", "url": "https://e2b.dev/pricing", - "date": "2024-10-01", - "value": "Usage-based pricing, free tier available" + "date": "2026-07-09", + "value": "Hobby plan free with one-time $100 usage credit; Pro $150/mo with 24-hour sessions and more concurrent sandboxes; Enterprise custom with BYOC/on-prem/self-hosted options. Usage billed per second (~$0.05/hr for a 1 vCPU sandbox)" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 80, @@ -380,7 +380,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "framework_integrations": { "score": 88, @@ -394,7 +394,7 @@ } ], "methodology": "Integration ecosystem assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "uptime": { "score": 84, @@ -408,7 +408,7 @@ } ], "methodology": "Uptime monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -422,11 +422,10 @@ "Multi-language support (Python, JavaScript, custom)" ], "limitations": [ - "Cloud-only, no self-hosting option currently", + "Self-hosting/BYOC (AWS and GCP) is enterprise-only and not self-serve; standard tiers are managed cloud only", "Usage-based pricing can be unpredictable for high-volume use", "Cold start latency (1-3s) can impact performance", "Limited to code execution use cases", - "Newer service with less production track record", "Network restrictions may limit some agent capabilities" ], "metadata": { @@ -440,7 +439,7 @@ "TypeScript", "Custom environments" ], - "deployment_type": "Managed cloud service", + "deployment_type": "Managed cloud service; enterprise BYOC (AWS/GCP), on-prem, and self-hosted options", "tool_support": [ "Agent framework integrations", "Custom code execution" @@ -454,7 +453,7 @@ "Custom" ], "github_org": "https://github.com/e2b-dev", - "pricing": "Free tier available, Paid plans from $150/month + usage" + "pricing": "Hobby free (one-time $100 usage credit); Pro $150/mo + per-second usage (~$0.05/hr per 1 vCPU sandbox); Enterprise custom (BYOC/on-prem/self-hosted)" }, "use_case_ratings": { "customer-support": { diff --git a/data/agents/flowise.json b/data/agents/flowise.json index 61261ef..8acd197 100644 --- a/data/agents/flowise.json +++ b/data/agents/flowise.json @@ -2,11 +2,11 @@ "id": "flowise", "type": "agent", "name": "Flowise", - "provider": "FlowiseAI", - "version": "2.x", - "last_evaluated": "2025-11-09", + "provider": "FlowiseAI (Workday since Aug 2025)", + "version": "3.1.3", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source low-code platform for building customized LLM orchestration flows and AI agents. Visual node-based editor for creating RAG pipelines, chatbots, and autonomous agents without extensive coding.", + "description": "Open-source low-code platform for LLM orchestration flows and AI agents, acquired by Workday August 2025. Visual node editor for RAG pipelines, chatbots, agents. SECURITY: three CVEs with confirmed in-the-wild exploitation - CVE-2025-59528 (CVSS 10.0 RCE via CustomMCP node, fixed in 3.0.6, exploited from April 2026, 12,000+ instances exposed), CVE-2025-8943 (CVSS 9.8 OS command RCE), CVE-2025-26319 (arbitrary file upload). Upgrade to 3.1.x; do not expose unauthenticated instances.", "website": "https://flowiseai.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Flow execution testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_chain_support": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "LLM chain testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "chatflow_management": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Chatflow capability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "vector_store_support": { "score": 86, @@ -66,7 +66,7 @@ } ], "methodology": "Vector store integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 71, @@ -80,7 +80,7 @@ } ], "methodology": "Error recovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "500ms-8s (varies)", @@ -94,15 +94,15 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 73, + "overall_score": 68, "criteria": { "api_authentication": { - "score": 76, + "score": 70, "confidence": "medium", "evidence": [ { @@ -110,10 +110,23 @@ "url": "https://docs.flowiseai.com/configuration/authorization", "date": "2024-10-01", "value": "API key authentication for chatflows and predictions" + }, + { + "source": "GHSA-3gcm-f6qx-ff7p (CVE-2025-59528)", + "url": "https://github.com/FlowiseAI/Flowise/security/advisories/GHSA-3gcm-f6qx-ff7p", + "date": "2026-07-09", + "value": "CVE-2025-59528: CVSS 10.0 code injection RCE via CustomMCP node config (fixed in 3.0.6, Sept 2025); first confirmed in-the-wild exploitation April 2026 with an estimated 12,000-15,000 internet-exposed instances" + }, + { + "source": "The Hacker News - Flowise exploitation", + "url": "https://thehackernews.com/2026/04/flowise-ai-agent-builder-under-active.html", + "date": "2026-07-09", + "value": "Third Flowise flaw with in-the-wild exploitation after CVE-2025-8943 (CVSS 9.8 OS command RCE) and CVE-2025-26319 (CVSS 8.9 arbitrary file upload)" } ], "methodology": "Authentication testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09", + "notes": "Score reduced 2026-07-09: three CVEs with confirmed in-the-wild exploitation (CVE-2025-59528, CVE-2025-8943, CVE-2025-26319); all patched in current 3.1.x releases but many exposed instances remain unpatched" }, "self_hosting": { "score": 88, @@ -127,7 +140,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_management": { "score": 74, @@ -141,7 +154,7 @@ } ], "methodology": "Credential security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 90, @@ -150,12 +163,12 @@ { "source": "GitHub", "url": "https://github.com/FlowiseAI/Flowise", - "date": "2024-10-20", - "value": "Apache 2.0 license, 32k+ stars, active development" + "date": "2026-07-09", + "value": "Apache 2.0 license, 54k+ stars, active development (now under Workday ownership)" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 68, @@ -169,7 +182,7 @@ } ], "methodology": "Data isolation assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +201,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 78, @@ -202,7 +215,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 90, @@ -216,7 +229,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 73, @@ -230,7 +243,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "embedding_storage": { "score": 76, @@ -244,7 +257,7 @@ } ], "methodology": "Data storage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +276,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "visual_interface": { "score": 90, @@ -277,7 +290,7 @@ } ], "methodology": "UI/UX assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 92, @@ -286,12 +299,12 @@ { "source": "GitHub", "url": "https://github.com/FlowiseAI/Flowise", - "date": "2024-10-20", - "value": "Apache 2.0, 32k+ stars, very active community" + "date": "2026-07-09", + "value": "Apache 2.0, 54k+ stars, very active community" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 82, @@ -305,7 +318,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "chatflow_export": { "score": 76, @@ -319,7 +332,7 @@ } ], "methodology": "Portability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +351,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 68, @@ -352,7 +365,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 92, @@ -366,7 +379,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 65, @@ -380,7 +393,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_deployment": { "score": 80, @@ -394,7 +407,7 @@ } ], "methodology": "API capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "marketplace": { "score": 78, @@ -408,14 +421,15 @@ } ], "methodology": "Template ecosystem assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } }, "strengths": [ "Extremely user-friendly visual interface for building AI agents", - "Open source (Apache 2.0) with very active community (32k+ stars)", + "Open source (Apache 2.0) with very active community (54k+ stars)", + "Backed by Workday since August 2025 acquisition", "Easy deployment with Docker, npm, or cloud platforms", "Built-in support for multiple vector stores and LLM providers", "Auto-generates API endpoints for chatflows", @@ -427,7 +441,8 @@ "Performance optimization requires technical knowledge", "Security features less mature than enterprise platforms", "Scaling to high-volume production requires additional infrastructure", - "Limited debugging capabilities for complex flows" + "Limited debugging capabilities for complex flows", + "Serious CVE history with confirmed in-the-wild exploitation: CVE-2025-59528 (CVSS 10.0 RCE, exploited April 2026), CVE-2025-8943 (CVSS 9.8 RCE) and CVE-2025-26319 (arbitrary file upload); upgrade to 3.0.6+/3.1.x and harden exposed deployments" ], "metadata": { "license": "Apache 2.0", @@ -450,12 +465,12 @@ "API integrations" ], "pricing_model": "Free open source (FlowiseAI Cloud managed service available)", - "github_stars": "42000+", + "github_stars": "54000+", "first_release": "2023", "vector_stores": "Pinecone, Weaviate, Qdrant, Chroma, Supabase, Postgres", "ui_technology": "React-based visual node editor", "acquisition": "Acquired by Workday in August 2025", - "version": "3.0.1+" + "version": "3.1.3 (June 25, 2026)" }, "use_case_ratings": { "customer-support": { @@ -508,6 +523,7 @@ "tags": [ "visual", "low-code", - "open-source" + "open-source", + "security-incidents" ] } diff --git a/data/agents/gemini-cli.json b/data/agents/gemini-cli.json index 8d61597..65966ef 100644 --- a/data/agents/gemini-cli.json +++ b/data/agents/gemini-cli.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Gemini CLI", "provider": "Google", - "version": "0.x (rolling release)", - "last_evaluated": "2026-06-10", + "version": "0.x (enterprise/API-key access only since 2026-06-18)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source terminal AI agent from Google that brings Gemini into the command line. Uses a ReAct loop with built-in tools, MCP server support, and Google Search grounding for coding, content generation, and task automation directly in the shell.", + "description": "Open-source terminal AI agent from Google that brings Gemini into the command line. Uses a ReAct loop with built-in tools, MCP server support, and Google Search grounding. Consumer/individual access (free tier and Google AI Pro/Ultra) ended 2026-06-18 as Google transitioned individuals to the closed-source Antigravity CLI ('agy'); Gemini CLI remains available to Gemini Code Assist Standard/Enterprise organizations and paid Gemini API key users.", "website": "https://github.com/google-gemini/gemini-cli", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Coding and shell task completion testing with large-context workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 84, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing across built-in and MCP tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 83, @@ -52,7 +52,7 @@ } ], "methodology": "Multi-step task decomposition testing via ReAct loop", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 74, @@ -66,7 +66,7 @@ } ], "methodology": "Session and context persistence evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 75, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling and retry behavior testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 65, @@ -94,7 +94,7 @@ } ], "methodology": "Multi-agent and scripting capability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Sandboxing options review and testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 76, @@ -127,7 +127,7 @@ } ], "methodology": "Access control configuration assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 75, @@ -141,7 +141,7 @@ } ], "methodology": "Injection surface review focusing on confirmation gates and web content handling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 72, @@ -155,7 +155,7 @@ } ], "methodology": "Data handling terms review across auth tiers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -169,7 +169,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy and terms review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 76, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance posture assessment per access tier", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 60, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 78, @@ -263,7 +263,7 @@ } ], "methodology": "Logging and telemetry capabilities assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 76, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 93, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 90, @@ -305,15 +305,15 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 74, + "overall_score": 70, "criteria": { "ease_of_integration": { - "score": 88, + "score": 80, "confidence": "high", "evidence": [ { @@ -321,10 +321,16 @@ "url": "https://blog.google/innovation-and-ai/technology/developers-tools/introducing-gemini-cli-open-source-ai-agent/", "date": "2025-06-25", "value": "Single npm install and Google account login; free tier with 60 requests/min and 1,000 requests/day" + }, + { + "source": "Google Developers Blog - Transitioning Gemini CLI to Antigravity CLI", + "url": "https://developers.googleblog.com/an-important-update-transitioning-gemini-cli-to-antigravity-cli/", + "date": "2026-06-18", + "value": "Simple Google-account onboarding no longer available: since 2026-06-18 access requires a Gemini Code Assist Standard/Enterprise license or a paid Gemini API key (score reduced 88 to 80)" } ], - "methodology": "Setup and onboarding complexity assessment", - "last_verified": "2026-06-10" + "methodology": "Setup and onboarding complexity assessment, updated for the post-2026-06-18 access model", + "last_verified": "2026-07-09" }, "scalability": { "score": 72, @@ -338,21 +344,21 @@ } ], "methodology": "Automation and rate-limit assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { - "score": 78, + "score": 72, "confidence": "medium", "evidence": [ { - "source": "Gemini CLI GitHub", - "url": "https://github.com/google-gemini/gemini-cli", - "date": "2026-06-01", - "value": "Generous free tier (1,000 requests/day); paid usage via metered Gemini API keys or Gemini Code Assist subscriptions" + "source": "Google Developers Blog - Transitioning Gemini CLI to Antigravity CLI", + "url": "https://developers.googleblog.com/an-important-update-transitioning-gemini-cli-to-antigravity-cli/", + "date": "2026-06-18", + "value": "Free consumer tier (1,000 requests/day) discontinued 2026-06-18; remaining usage is via metered Gemini API keys or flat Gemini Code Assist Standard/Enterprise subscriptions (score reduced 78 to 72)" } ], - "methodology": "Pricing model analysis", - "last_verified": "2026-06-10" + "methodology": "Pricing model analysis updated for removal of the free consumer tier", + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 70, @@ -366,21 +372,21 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { - "score": 62, - "confidence": "medium", + "score": 58, + "confidence": "high", "evidence": [ { - "source": "Digital Applied - Gemini CLI to Antigravity CLI Migration Guide", - "url": "https://www.digitalapplied.com/blog/gemini-cli-to-antigravity-cli-migration-june-18-2026-guide", - "date": "2026-06-05", - "value": "Third-party reporting indicates consumer access to Gemini CLI ends 2026-06-18 as Google migrates individual users to Antigravity CLI; enterprise Gemini Code Assist users retain access" + "source": "Google Developers Blog - Transitioning Gemini CLI to Antigravity CLI", + "url": "https://developers.googleblog.com/an-important-update-transitioning-gemini-cli-to-antigravity-cli/", + "date": "2026-06-18", + "value": "Officially confirmed and executed: on 2026-06-18 Gemini CLI (and Gemini Code Assist IDE extensions) stopped serving Google AI Pro/Ultra and free individual users; individuals are directed to the closed-source Go-based Antigravity CLI ('agy'), which lacks 1:1 feature parity and uses weekly compute caps. Organizations on Gemini Code Assist Standard/Enterprise and paid Gemini API key users retain access" } ], - "methodology": "Product continuity assessment based on third-party migration reporting; not yet confirmed in official Google release notes", - "last_verified": "2026-06-10" + "methodology": "Product continuity assessment; the consumer shutdown is now confirmed by Google's official announcement, so score lowered 62 to 58 and confidence raised to high — the product continues but only for enterprise/API-key audiences", + "last_verified": "2026-07-09" } } } @@ -403,26 +409,25 @@ "notes": "Useful for drafting docs and content in the terminal with multimodal generation via extensions" }, "education": { - "overall": 75, - "notes": "Free tier makes it accessible for learning, but consumer access continuity is uncertain" + "overall": 60, + "notes": "No longer accessible to individual learners since the 2026-06-18 consumer shutdown; requires an enterprise Code Assist license or paid API key" } }, "best_for": [ - "Developers wanting a free, open-source AI agent in the terminal", - "Teams already on Google's Gemini ecosystem and Gemini Code Assist", + "Organizations on Gemini Code Assist Standard/Enterprise wanting an open-source terminal agent", + "Teams already on Google's Gemini ecosystem with paid Gemini API keys", "Scripted automation via non-interactive mode and MCP servers", "Quick coding, debugging, and research tasks with Search grounding" ], "strengths": [ - "Open source (Apache 2.0) with very large and active community", - "Generous free tier: 60 requests/min and 1,000 requests/day", + "Open source (Apache 2.0) with a very large community (6,000+ merged community PRs)", "Strong safety options: Docker/Seatbelt sandboxing and per-tool confirmation prompts", "ReAct loop with built-in tools, MCP support, and Google Search grounding", - "1M token context window via Gemini 2.5 Pro for large codebases" + "1M token context window via Gemini Pro models for large codebases" ], "limitations": [ - "Third-party reports indicate consumer access ends 2026-06-18 with migration to Antigravity CLI; only enterprise Gemini Code Assist retains access (unconfirmed officially, medium confidence)", - "Free personal tier may use prompts and code for Google product improvement", + "Consumer/individual access ended 2026-06-18 (officially confirmed): free tier and Google AI Pro/Ultra users were cut off and directed to the closed-source Antigravity CLI, which lacks full feature parity and uses weekly compute caps; only Gemini Code Assist Standard/Enterprise orgs and paid API key users retain Gemini CLI access", + "Historical free personal tier could use prompts and code for Google product improvement", "No local model backend; inference requires Google's cloud Gemini API", "Single-agent design with no native multi-agent orchestration", "Rate limits and model fallback (Pro to Flash) can degrade quality under load" @@ -446,8 +451,8 @@ "Extensions" ], "first_release": "2025-06-25", - "pricing": "Free tier (60 req/min, 1,000 req/day with personal Google account); paid via Gemini API keys or Gemini Code Assist Standard/Enterprise", - "product_status_note": "Consumer access reportedly ending 2026-06-18 in favor of Antigravity CLI per third-party reporting; enterprise access continues via Gemini Code Assist" + "pricing": "Free consumer tier discontinued 2026-06-18; access via metered Gemini API keys or Gemini Code Assist Standard/Enterprise subscriptions", + "product_status_note": "Officially confirmed by Google (developers.googleblog.com): consumer/individual access ended 2026-06-18 in favor of the closed-source Antigravity CLI ('agy'); enterprise access continues via Gemini Code Assist Standard/Enterprise and paid Gemini API keys" }, "tags": [ "cli", diff --git a/data/agents/github-copilot-coding-agent.json b/data/agents/github-copilot-coding-agent.json index 3358efd..6b90c47 100644 --- a/data/agents/github-copilot-coding-agent.json +++ b/data/agents/github-copilot-coding-agent.json @@ -4,7 +4,7 @@ "name": "GitHub Copilot Coding Agent", "provider": "GitHub (Microsoft)", "version": "GA (2025-09)", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Autonomous background coding agent built into GitHub. Assign it a GitHub issue or prompt and it works in an ephemeral GitHub Actions sandbox, then opens a draft pull request for human review. Distinct from Copilot's interactive IDE agent mode.", "website": "https://docs.github.com/en/copilot/concepts/agents/coding-agent/about-coding-agent", @@ -24,7 +24,7 @@ } ], "methodology": "Task scope analysis and PR outcome review on representative issues", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 83, @@ -38,7 +38,7 @@ } ], "methodology": "Tooling and environment reliability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Multi-step task execution and iteration testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 72, @@ -66,7 +66,7 @@ } ], "methodology": "Cross-session context persistence evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 75, @@ -80,7 +80,7 @@ } ], "methodology": "Failure iteration and review-feedback loop testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 64, @@ -94,7 +94,7 @@ } ], "methodology": "Concurrency and orchestration capability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Sandbox and network isolation architecture review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 86, @@ -127,7 +127,7 @@ } ], "methodology": "Permission boundary and branch protection review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 78, @@ -141,7 +141,7 @@ } ], "methodology": "Injection mitigation review against documented threat model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 82, @@ -155,7 +155,7 @@ } ], "methodology": "Session and tenant isolation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 35, @@ -169,7 +169,7 @@ } ], "methodology": "Source availability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data handling and retention terms review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 84, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance program and DPA assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 74, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis across model backends", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 25, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 86, @@ -263,7 +263,7 @@ } ], "methodology": "Session log and commit trail assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 30, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 78, @@ -305,7 +305,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Onboarding and integration assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 84, @@ -338,7 +338,7 @@ } ], "methodology": "Parallelism and quota analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 68, @@ -349,10 +349,16 @@ "url": "https://github.blog/news-insights/company-news/github-copilot-is-moving-to-usage-based-billing/", "date": "2026-06-01", "value": "Premium-request billing since 2025-06-18; from 2026-06-01 GitHub is transitioning to token-based 'GitHub AI Credits', making per-task costs harder to forecast during the changeover" + }, + { + "source": "GitHub Docs - Usage-based billing", + "url": "https://docs.github.com/en/copilot/concepts/billing/usage-based-billing-for-organizations-and-enterprises", + "date": "2026-07-09", + "value": "AI Credits billing is now live (since 2026-06-01): 1 credit = $0.01, metered on token consumption at listed model rates. Pro includes $10/mo and Pro+ $39/mo in credits; monthly plans auto-migrated while annual plans keep premium requests until expiry; Business/Enterprise get boosted included credits Jun-Sep 2026 plus budget controls at enterprise, cost-center, and user levels" } ], - "methodology": "Pricing model analysis including billing model transition", - "last_verified": "2026-06-10" + "methodology": "Pricing model analysis including the completed billing model transition", + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 82, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring and audit features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 86, @@ -380,7 +386,7 @@ } ], "methodology": "Product maturity and availability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -419,7 +425,7 @@ ], "limitations": [ "Proprietary and cloud-only; no self-hosted runner support for the agent", - "Billing complexity: premium requests since 2025-06-18 and a transition to token-based GitHub AI Credits beginning 2026-06-01 make costs harder to predict", + "Billing complexity: token-based GitHub AI Credits (live since 2026-06-01, 1 credit = $0.01) replaced premium requests for monthly plans, while annual plans stay on legacy premium requests until expiry, making per-task costs harder to predict during the mixed-model period", "Requires a paid Copilot plan (Pro, Pro+, Business, or Enterprise); not in Copilot Free", "Best on well-scoped tasks; struggles with large cross-repo or ambiguous refactors", "Consumes GitHub Actions minutes in addition to premium requests/credits", @@ -444,7 +450,7 @@ "Draft PR creation" ], "first_release": "Preview May 2025; GA September 2025", - "pricing": "Copilot Pro $10/mo, Pro+ $39/mo, Business/Enterprise per-seat; agent usage billed via premium requests (since 2025-06-18), transitioning to token-based GitHub AI Credits from 2026-06-01" + "pricing": "Copilot Pro $10/mo (includes $10 AI Credits), Pro+ $39/mo (includes $39 AI Credits), Business/Enterprise per-seat with pooled credits; token-based GitHub AI Credits billing live since 2026-06-01 (1 credit = $0.01); annual plans remain on legacy premium requests until renewal" }, "tags": [ "coding-agent", diff --git a/data/agents/glean-ai.json b/data/agents/glean-ai.json index fb86ff9..4731c05 100644 --- a/data/agents/glean-ai.json +++ b/data/agents/glean-ai.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Glean AI", "provider": "Glean Technologies Inc.", - "version": "2025.1", - "last_evaluated": "2025-01-14", + "version": "Current (SaaS)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Enterprise AI platform for work that unifies information across business tools and applications. Provides intelligent search, knowledge discovery, and AI assistants that understand organizational context for improved productivity.", + "description": "Enterprise 'Work AI' platform that unifies information across business tools and applications. Has expanded well beyond enterprise search into AI assistants and autonomous agents that execute tasks across the business (Glean Assistant, Glean Agents, plus Glean Protect for agent governance/security). Raised a $150M Series F at a $7.2B valuation (June 2025) and crossed $300M ARR by May 2026; customers include Dell, Workday, and Palo Alto Networks.", "website": "https://www.glean.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Search quality testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "knowledge_synthesis": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Knowledge synthesis testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "integration_coverage": { "score": 92, @@ -52,7 +52,7 @@ } ], "methodology": "Integration coverage review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "response_quality": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Response quality assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "personalization": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Personalization testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -94,12 +94,12 @@ { "source": "Glean Security", "url": "https://www.glean.com/security", - "date": "2025-01-10", - "value": "SOC 2 Type II, ISO 27001, GDPR compliant" + "date": "2026-07-09", + "value": "SOC 2 Type II, ISO/IEC 27001, ISO/IEC 42001:2023 (AI management systems), HIPAA, GDPR compliant; Trust Center at trust.glean.com" } ], "methodology": "Security certification review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "permission_inheritance": { "score": 95, @@ -113,7 +113,7 @@ } ], "methodology": "Permission model review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_encryption": { "score": 92, @@ -127,7 +127,7 @@ } ], "methodology": "Encryption review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "sso_integration": { "score": 92, @@ -141,7 +141,7 @@ } ], "methodology": "SSO integration review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "zero_trust_architecture": { "score": 88, @@ -155,7 +155,7 @@ } ], "methodology": "Architecture review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "GDPR compliance review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 92, @@ -188,7 +188,7 @@ } ], "methodology": "Data isolation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_residency": { "score": 88, @@ -202,7 +202,7 @@ } ], "methodology": "Data residency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "no_training_on_data": { "score": 90, @@ -216,7 +216,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "source_attribution": { "score": 92, @@ -249,7 +249,7 @@ } ], "methodology": "Attribution testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "activity_logging": { "score": 85, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capability review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "model_transparency": { "score": 78, @@ -277,7 +277,7 @@ } ], "methodology": "Model transparency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Deployment assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "user_adoption": { "score": 90, @@ -310,7 +310,7 @@ } ], "methodology": "User experience review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "scalability": { "score": 88, @@ -324,7 +324,7 @@ } ], "methodology": "Scalability assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -338,7 +338,7 @@ } ], "methodology": "Support quality review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "api_extensibility": { "score": 82, @@ -352,7 +352,7 @@ } ], "methodology": "API capability review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } @@ -410,7 +410,7 @@ "Respects existing permission models", "100+ pre-built connectors", "AI responses grounded in organizational knowledge", - "Strong enterprise security (SOC 2, ISO 27001)", + "Strong enterprise security (SOC 2 Type II, ISO 27001, ISO 42001, HIPAA)", "No customer data used for training" ], "limitations": [ @@ -427,10 +427,11 @@ "deployment_type": "SaaS", "connectors": "100+", "pricing": "Enterprise pricing - contact sales", - "certifications": ["SOC 2 Type II", "ISO 27001", "GDPR"], + "certifications": ["SOC 2 Type II", "ISO 27001", "ISO/IEC 42001:2023", "HIPAA", "GDPR"], "founded": "2019", "headquarters": "Palo Alto, CA", - "funding": "Series D ($200M at $2.2B valuation)" + "funding": "Series F ($150M at $7.2B valuation, June 2025, led by Wellington Management)", + "annual_revenue": "$300M+ ARR (May 2026)" }, "tags": ["enterprise", "knowledge-management", "search", "productivity"] } diff --git a/data/agents/google-adk.json b/data/agents/google-adk.json index 9c00045..7cdfff2 100644 --- a/data/agents/google-adk.json +++ b/data/agents/google-adk.json @@ -3,8 +3,8 @@ "type": "agent", "name": "Google Agent Development Kit (ADK)", "provider": "Google", - "version": "2.0", - "last_evaluated": "2026-06-10", + "version": "2.x", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Open-source, code-first framework for building, evaluating, and deploying AI agents. Supports workflow agents, multi-agent hierarchies, built-in evaluation, and deployment to Vertex AI Agent Engine. Underlies Google's broader agent stack and is Gemini-optimized but model-agnostic.", "website": "https://google.github.io/adk-docs/", @@ -24,7 +24,7 @@ } ], "methodology": "Review of built-in evaluation tooling and reported agent benchmark workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing across function, OpenAPI, and MCP tool types", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "Assessment of workflow agent primitives and graph-based orchestration in ADK 2.0", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Memory and session service architecture evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 78, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling pattern review and community issue analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 90, @@ -94,7 +94,7 @@ } ], "methodology": "Multi-agent hierarchy and delegation capability testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of tool execution paths", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 78, @@ -127,7 +127,7 @@ } ], "methodology": "Access control and authentication capability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 74, @@ -141,7 +141,7 @@ } ], "methodology": "Review of documented safety patterns and guardrail mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 80, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture and session isolation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 93, @@ -169,7 +169,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review across deployment options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 83, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment for framework and managed deployments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 75, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis of model and tool integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 86, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment including local model support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 84, @@ -263,7 +263,7 @@ } ], "methodology": "Tracing and observability capabilities assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 78, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 93, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 87, @@ -302,10 +302,16 @@ "url": "https://github.com/google/adk-python", "date": "2026-06-10", "value": "Active development with frequent releases, multi-language SDK expansion, and growing contributor base since April 2025 launch" + }, + { + "source": "adk-python GitHub Releases", + "url": "https://github.com/google/adk-python/releases", + "date": "2026-07-09", + "value": "Roughly bi-weekly cadence continues: v2.4.0 released 2026-07-07; 1.x maintenance line still receiving patches (v1.36.1 on 2026-07-06)" } ], "methodology": "Community engagement and release cadence analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +330,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 86, @@ -338,7 +344,7 @@ } ], "methodology": "Deployment and scaling options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 85, @@ -352,7 +358,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 80, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 84, @@ -380,7 +386,7 @@ } ], "methodology": "Production readiness and release maturity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -460,6 +466,8 @@ "Third-party library tools" ], "first_release": "2025 (announced Cloud Next 2025-04-09; Python v1.0 May 2025; ADK 2.0 GA 2026-05-19)", + "latest_version": "v2.4.0 (2026-07-07); 1.x maintenance line at v1.36.1 (2026-07-06)", + "documentation": "https://google.github.io/adk-docs/ (docs also served at https://adk.dev/)", "pricing": "Free framework (Apache 2.0); paid when using Vertex AI Agent Engine or Gemini Enterprise Agent Platform plus model API costs" }, "tags": [ diff --git a/data/agents/google-agent-builder.json b/data/agents/google-agent-builder.json index 550dfdf..bfb4e73 100644 --- a/data/agents/google-agent-builder.json +++ b/data/agents/google-agent-builder.json @@ -4,9 +4,9 @@ "name": "Gemini Enterprise Agent Platform (formerly Vertex AI Agent Builder)", "provider": "Google Cloud", "version": "2026", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "REBRANDED: at Cloud Next (April 2026) Vertex AI Agent Builder became the Gemini Enterprise Agent Platform (APIs unchanged), following Agentspace's absorption into Gemini Enterprise in Oct 2025. Google Cloud's managed platform for building conversational AI agents and search apps with no-code/low-code options, enterprise search, and grounding.", + "description": "REBRANDED: at Cloud Next (April 2026) Vertex AI Agent Builder became the Gemini Enterprise Agent Platform (APIs unchanged), following Agentspace's absorption into Gemini Enterprise in Oct 2025. Console migration completed ~May 21, 2026: Vertex AI branding is gone, with all Vertex AI capabilities under the new name and no breaking changes. Google Cloud's managed platform for conversational AI agents and search apps with no-code/low-code options, enterprise search, and grounding.", "website": "https://cloud.google.com/products/gemini-enterprise-agent-platform", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on Gemini model performance and managed service reliability", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 89, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 86, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 87, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "grounding_accuracy": { "score": 91, @@ -94,7 +94,7 @@ } ], "methodology": "Grounding capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 95, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 87, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 94, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "enterprise_security": { "score": 96, @@ -169,7 +169,7 @@ } ], "methodology": "Enterprise security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 94, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 90, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "regional_deployment": { "score": 90, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 86, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 78, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "managed_service_sla": { "score": 90, @@ -291,7 +291,7 @@ } ], "methodology": "SLA review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "proprietary_service": { "score": 68, @@ -305,7 +305,7 @@ } ], "methodology": "Transparency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 95, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 84, @@ -352,7 +352,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09", + "last_verified": "2026-07-09", "notes": "Costs can vary with grounding and search usage" }, "monitoring_capabilities": { @@ -367,7 +367,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 92, @@ -384,10 +384,16 @@ "url": "https://cloud.google.com/products/gemini-enterprise-agent-platform", "date": "2026-06-10", "value": "Rebranded at Cloud Next April 2026 from Vertex AI Agent Builder to Gemini Enterprise Agent Platform; APIs unchanged. Agentspace was absorbed into Gemini Enterprise in Oct 2025" + }, + { + "source": "Gemini Enterprise Agent Platform (console migration)", + "url": "https://cloud.google.com/products/gemini-enterprise-agent-platform", + "date": "2026-07-09", + "value": "Console migration completed ~May 21, 2026: Vertex AI branding removed from Google Cloud Console; SDKs, billing, and APIs migrated with no breaking changes for existing deployments" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -402,19 +408,20 @@ ], "limitations": [ "Google Cloud vendor lock-in with proprietary service", - "Limited to Gemini and PaLM models", + "Primarily Gemini-centric (Model Garden offers partner/open models, but the platform is optimized for Gemini)", "Higher costs for grounding and search features", "Less code-level flexibility than open frameworks", "Requires Google Cloud expertise for optimization", "Not open source, limited customization of orchestration", - "Repeated rebranding (Agentspace into Gemini Enterprise Oct 2025; Agent Builder renamed April 2026) makes older docs/links stale, though APIs are unchanged" + "Repeated rebranding (Agentspace into Gemini Enterprise Oct 2025; Agent Builder renamed April 2026; Vertex AI branding removed from console May 2026) makes older docs/links stale, though APIs are unchanged" ], "metadata": { "license": "Proprietary (Google Cloud)", "supported_models": [ - "Gemini Pro", - "Gemini Ultra", - "PaLM 2" + "Gemini 3 family", + "Gemini 2.5 family", + "Partner and open models via Model Garden", + "PaLM 2 (retired)" ], "programming_languages": [ "Google Cloud SDK (Python, Java, Node.js, etc.)", diff --git a/data/agents/google-dialogflow.json b/data/agents/google-dialogflow.json index 0a8cdb7..5607154 100644 --- a/data/agents/google-dialogflow.json +++ b/data/agents/google-dialogflow.json @@ -1,13 +1,13 @@ { "id": "google-dialogflow", "type": "agent", - "name": "Google Dialogflow CX", + "name": "Google Conversational Agents (Dialogflow CX)", "provider": "Google Cloud", - "version": "CX", - "last_evaluated": "2025-11-09", + "version": "Conversational Agents (CX)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Google's advanced conversational AI platform for building sophisticated virtual agents with visual flow design, state management, and enterprise features. Successor to Dialogflow ES with enhanced capabilities for complex conversations.", - "website": "https://cloud.google.com/dialogflow/cx/docs", + "description": "REBRANDED: Dialogflow CX is now sold as Conversational Agents on Google Cloud - the standalone CX console was deprecated October 31, 2025, and users are auto-routed to the Conversational Agents console, where generative playbooks and LLM-driven fulfillment sit alongside classic deterministic flows. Google's advanced conversational AI platform for building virtual agents with visual flow design, state management, and enterprise features. Successor to Dialogflow ES.", + "website": "https://cloud.google.com/products/conversational-agents", "trust_vector": { "performance_reliability": { "overall_score": 89, @@ -24,7 +24,7 @@ } ], "methodology": "Intent classification benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversation_flow": { "score": 94, @@ -38,7 +38,7 @@ } ], "methodology": "Complex conversation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "entity_extraction": { "score": 91, @@ -52,7 +52,7 @@ } ], "methodology": "Entity extraction testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "state_management": { "score": 90, @@ -66,7 +66,7 @@ } ], "methodology": "State persistence testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_language": { "score": 88, @@ -80,7 +80,7 @@ } ], "methodology": "Multi-language testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "200-400ms average", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 91, @@ -127,7 +127,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "vpc_integration": { "score": 89, @@ -141,7 +141,7 @@ } ], "methodology": "Network security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 92, @@ -155,7 +155,7 @@ } ], "methodology": "Audit capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance": { "score": 91, @@ -169,7 +169,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 90, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "hipaa_compliance": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Healthcare compliance review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_redaction": { "score": 89, @@ -230,7 +230,7 @@ } ], "methodology": "PII protection testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_residency": { "score": 88, @@ -244,7 +244,7 @@ } ], "methodology": "Data residency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -260,10 +260,16 @@ "url": "https://cloud.google.com/dialogflow/cx/docs", "date": "2024-10-20", "value": "Comprehensive documentation with best practices" + }, + { + "source": "Dialogflow release notes (Conversational Agents rebrand)", + "url": "https://docs.cloud.google.com/dialogflow/docs/release-notes", + "date": "2026-07-09", + "value": "Dialogflow CX console deprecated October 31, 2025; users automatically routed to the Conversational Agents console; product rebranded Conversational Agents with generative playbooks promoted" } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "visual_design_tools": { "score": 90, @@ -277,7 +283,7 @@ } ], "methodology": "Design tools assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "analytics": { "score": 86, @@ -291,7 +297,7 @@ } ], "methodology": "Analytics capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "testing_simulator": { "score": 84, @@ -305,7 +311,7 @@ } ], "methodology": "Testing features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +330,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 93, @@ -338,7 +344,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 84, @@ -349,10 +355,16 @@ "url": "https://cloud.google.com/dialogflow/cx/pricing", "date": "2024-10-01", "value": "Session-based pricing, $0.007 per session (first 1000 free)" + }, + { + "source": "Conversational Agents pricing", + "url": "https://cloud.google.com/products/conversational-agents/pricing", + "date": "2026-07-09", + "value": "Consumption-based pricing under Conversational Agents branding: billed per request for chat agents and per second of audio for voice agents; new customers receive $600 in credits (12 months)" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 89, @@ -366,7 +378,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "versioning": { "score": 87, @@ -380,7 +392,7 @@ } ], "methodology": "Version control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "integrations": { "score": 88, @@ -394,7 +406,7 @@ } ], "methodology": "Integration options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -408,6 +420,7 @@ "Strong enterprise features (HIPAA, telephony, contact center integration)" ], "limitations": [ + "Rebranding to Conversational Agents (CX console deprecated Oct 31, 2025) makes older docs, tutorials, and links stale, though the underlying platform is unchanged", "Steep learning curve for complex flow design", "Higher cost compared to Dialogflow ES or open-source alternatives", "Google Cloud Platform lock-in and dependencies", @@ -446,7 +459,7 @@ "Audit logging" ], "telephony_support": true, - "pricing": "Dialogflow ES: $0.0025/text request (7.5M messages/month free); Dialogflow CX: $0.0050-0.0065/query or $20 per 100 sessions. Enterprise support from $10k/month. New customers get $600 credit (12 months)" + "pricing": "Now billed as Conversational Agents: consumption-based, per request for chat agents and per second of audio for voice agents (see cloud.google.com/products/conversational-agents/pricing). Legacy references: Dialogflow ES $0.0025/text request; CX $0.0050-0.0065/query or $20 per 100 sessions. New customers get $600 credit (12 months)" }, "use_case_ratings": { "customer-support": { @@ -497,6 +510,7 @@ "Contact centers implementing intelligent virtual agents" ], "tags": [ - "google" + "google", + "rebranded" ] } diff --git a/data/agents/google-jules.json b/data/agents/google-jules.json index 2985aba..58af476 100644 --- a/data/agents/google-jules.json +++ b/data/agents/google-jules.json @@ -4,9 +4,9 @@ "name": "Google Jules", "provider": "Google", "version": "GA (2025-08-06)", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Asynchronous autonomous coding agent from Google. Jules clones a repository into an isolated Google Cloud VM, plans and writes code in the background, runs tests, and opens pull requests for human review. Powered by Gemini 2.5/3 models.", + "description": "Asynchronous autonomous coding agent from Google. Jules clones a repository into an isolated Google Cloud VM, plans and writes code in the background, runs tests, and opens pull requests for human review. Powered by Gemini 2.5 Pro on the free tier and Gemini 3 Pro on paid Google AI Pro/Ultra tiers.", "website": "https://jules.google/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Assessment of reported task outcomes and hands-on PR quality review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 79, @@ -38,7 +38,7 @@ } ], "methodology": "VM tooling and environment setup reliability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Plan generation and decomposition quality review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 76, @@ -66,7 +66,7 @@ } ], "methodology": "Cross-task context persistence evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 75, @@ -80,7 +80,7 @@ } ], "methodology": "Failure handling and iteration behavior testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 62, @@ -94,7 +94,7 @@ } ], "methodology": "Concurrency and orchestration capability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Execution isolation architecture review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 78, @@ -127,7 +127,7 @@ } ], "methodology": "Repository permission model assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 72, @@ -141,7 +141,7 @@ } ], "methodology": "Injection surface review focusing on autonomy boundaries and human gates", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 80, @@ -155,7 +155,7 @@ } ], "methodology": "Tenant and task isolation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 35, @@ -169,7 +169,7 @@ } ], "methodology": "Source availability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy terms and data handling review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 74, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance posture assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 65, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 25, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 84, @@ -263,7 +263,7 @@ } ], "methodology": "Execution visibility assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 82, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 30, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 70, @@ -305,7 +305,7 @@ } ], "methodology": "Release cadence and community engagement analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Onboarding and integration assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 75, @@ -338,21 +338,21 @@ } ], "methodology": "Concurrency and quota analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 84, "confidence": "high", "evidence": [ { - "source": "Jules Pricing", - "url": "https://jules.google/", - "date": "2026-06-01", - "value": "Flat subscription tiers: Free (15 tasks/day, 3 concurrent), Google AI Pro $19.99/mo (~100 tasks/day), AI Ultra (~300 tasks/day)" + "source": "Jules usage limits documentation", + "url": "https://jules.google/docs/usage-limits/", + "date": "2026-07-09", + "value": "Flat subscription tiers confirmed: Free (15 tasks/day, 3 concurrent, Gemini 2.5 Pro), Google AI Pro $19.99/mo (100 tasks/day, 15 concurrent, Gemini 3 Pro), AI Ultra (300 tasks/day, 60 concurrent); Ultra entry price cut to $99.99/mo (5x Pro) at Google I/O 2026, with a $200/mo top tier (20x)" } ], "methodology": "Pricing model analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 72, @@ -366,7 +366,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 80, @@ -380,7 +380,7 @@ } ], "methodology": "Product maturity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -423,13 +423,14 @@ "Daily task limits even on paid tiers (~100/day Pro, ~300/day Ultra)", "Locked to Gemini models with no model choice", "GitHub-focused; weaker support for other source forges", - "Lacks enterprise compliance attestations and audit tooling of Google Cloud products" + "Lacks enterprise compliance attestations and audit tooling of Google Cloud products", + "Paid Jules tiers are currently restricted to individual Google Accounts (@gmail.com); no workspace/organizational upgrade path yet" ], "metadata": { "license": "Proprietary", "supported_models": [ - "Gemini 2.5 Pro", - "Gemini 3 (per Google updates)" + "Gemini 2.5 Pro (free tier)", + "Gemini 3 Pro (Google AI Pro/Ultra tiers)" ], "programming_languages": [ "Most major languages (Python, JavaScript/TypeScript, Java, Go, Rust, and more)" @@ -443,7 +444,7 @@ "Jules API and Jules Tools CLI" ], "first_release": "Public beta May 2025 (Google I/O); GA 2025-08-06", - "pricing": "Free: 15 tasks/day, 3 concurrent; Google AI Pro $19.99/mo (~100 tasks/day, 15 concurrent); Google AI Ultra (~300 tasks/day, 60 concurrent)" + "pricing": "Free: 15 tasks/day, 3 concurrent; Google AI Pro $19.99/mo (100 tasks/day, 15 concurrent); Google AI Ultra $99.99/mo entry or $200/mo top tier (300 tasks/day, 60 concurrent); paid tiers limited to individual @gmail accounts" }, "tags": [ "coding-agent", diff --git a/data/agents/haystack.json b/data/agents/haystack.json index 6a71367..71e535c 100644 --- a/data/agents/haystack.json +++ b/data/agents/haystack.json @@ -4,9 +4,9 @@ "name": "Haystack", "provider": "deepset", "version": "2.x", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source NLP framework for building production-ready LLM applications, RAG pipelines, and semantic search systems. Modular architecture with pre-built components for document processing, retrieval, and generation.", + "description": "Open-source AI orchestration framework from deepset for building production-ready LLM applications, RAG pipelines, agent workflows, and semantic search systems. Modular architecture with pre-built components for document processing, retrieval, and generation, plus Agent components added in the 2.x line. Actively maintained (2.31.0 released July 2026) with commercial support via Haystack Enterprise and the deepset AI Platform.", "website": "https://haystack.deepset.ai/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "RAG pipeline benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "document_retrieval": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Retrieval accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pipeline_flexibility": { "score": 92, @@ -52,7 +52,7 @@ } ], "methodology": "Architecture assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_integration": { "score": 86, @@ -66,7 +66,7 @@ } ], "methodology": "LLM integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "document_processing": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Document processing testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Varies by pipeline (500ms-5s)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_security": { "score": 72, @@ -127,7 +127,7 @@ } ], "methodology": "API security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_privacy": { "score": 85, @@ -141,7 +141,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 94, @@ -152,10 +152,16 @@ "url": "https://github.com/deepset-ai/haystack", "date": "2024-10-20", "value": "Apache 2.0 license, 17k+ stars, active development" + }, + { + "source": "GitHub", + "url": "https://github.com/deepset-ai/haystack", + "date": "2026-07-09", + "value": "Apache 2.0 license, 25.9k stars, active development; latest release v2.31.0 (2026-07-08)" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "document_storage_security": { "score": 68, @@ -169,7 +175,7 @@ } ], "methodology": "Storage security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +194,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 84, @@ -202,7 +208,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 92, @@ -216,7 +222,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 75, @@ -230,7 +236,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "no_telemetry": { "score": 90, @@ -244,7 +250,7 @@ } ], "methodology": "Telemetry assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +269,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 95, @@ -272,12 +278,12 @@ { "source": "GitHub", "url": "https://github.com/deepset-ai/haystack", - "date": "2024-10-20", - "value": "Apache 2.0, 17k+ stars, transparent development" + "date": "2026-07-09", + "value": "Apache 2.0, 25.9k stars, transparent development with frequent releases (v2.31.0, 2026-07-08)" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pipeline_traceability": { "score": 84, @@ -291,7 +297,7 @@ } ], "methodology": "Traceability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 87, @@ -305,7 +311,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +330,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 78, @@ -338,7 +344,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 92, @@ -352,7 +358,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 75, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 79, @@ -380,7 +386,7 @@ } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "modular_architecture": { "score": 91, @@ -394,7 +400,7 @@ } ], "methodology": "Architecture assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -402,7 +408,7 @@ "strengths": [ "Open-source (Apache 2.0) specialized for RAG and semantic search", "Modular architecture with 100+ pre-built integrations", - "Excellent documentation and active community (17k+ stars)", + "Excellent documentation and active community (25k+ stars)", "Supports multiple LLM providers and local models", "Production-ready with REST API and container deployment", "Strong document retrieval and processing capabilities" @@ -412,7 +418,7 @@ "Limited built-in monitoring and observability features", "Setup complexity higher than managed services", "Performance tuning requires deep understanding", - "Limited agent-like autonomous behavior capabilities", + "Agent components arrived later in the 2.x line; agent tooling younger than dedicated agent frameworks", "Document store choice affects performance and cost significantly" ], "metadata": { @@ -427,16 +433,17 @@ "programming_languages": [ "Python" ], - "deployment_type": "Self-hosted (Docker, Kubernetes) or deepset Cloud", + "deployment_type": "Self-hosted (Docker, Kubernetes) or managed via Haystack Enterprise Platform / deepset AI Platform", "tool_support": [ "Document stores", "Vector DBs", "Embedding models", "LLMs" ], - "pricing_model": "Free open source (deepset Cloud managed service available)", - "github_stars": "21400+", + "pricing_model": "Free open source (paid Haystack Enterprise Starter support and Haystack Enterprise Platform managed/self-hosted offerings from deepset)", + "github_stars": "25900+", "first_release": "2019", + "latest_version": "2.31.0 (July 8, 2026)", "supported_document_stores": "Elasticsearch, OpenSearch, Weaviate, Pinecone, Qdrant, Milvus", "use_case_focus": "RAG, semantic search, question answering", "version": "2.x", diff --git a/data/agents/ibm-watson-assistant.json b/data/agents/ibm-watson-assistant.json index 49d96a0..f4f283b 100644 --- a/data/agents/ibm-watson-assistant.json +++ b/data/agents/ibm-watson-assistant.json @@ -1,13 +1,13 @@ { "id": "ibm-watson-assistant", "type": "agent", - "name": "IBM Watson Assistant", + "name": "IBM watsonx Assistant (now part of watsonx Orchestrate)", "provider": "IBM", - "version": "4.x", - "last_evaluated": "2025-11-09", + "version": "watsonx Orchestrate (2026)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "IBM's enterprise-grade conversational AI platform powered by Watson AI. Combines natural language understanding, dialog management, and integration capabilities for building sophisticated virtual agents across industries.", - "website": "https://www.ibm.com/cloud/watson-assistant", + "description": "CONSOLIDATED: Watson Assistant was rebranded watsonx Assistant (2023) and folded into IBM watsonx Orchestrate - the Assistant pricing page now redirects to Orchestrate, and IBM positions new implementations on Orchestrate, its agentic AI platform (next generation announced at Think 2026, May 2026, as a multi-agent orchestration and governance control plane). Existing Assistant deployments continue to run. Combines NLU, dialog skills, and integrations for enterprise virtual agents.", + "website": "https://www.ibm.com/products/watsonx-orchestrate", "trust_vector": { "performance_reliability": { "overall_score": 86, @@ -24,7 +24,7 @@ } ], "methodology": "Intent accuracy benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "entity_extraction": { "score": 87, @@ -38,7 +38,7 @@ } ], "methodology": "Entity extraction testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "dialog_management": { "score": 89, @@ -52,7 +52,7 @@ } ], "methodology": "Complex conversation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "context_awareness": { "score": 85, @@ -66,7 +66,7 @@ } ], "methodology": "Context retention testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_integration": { "score": 84, @@ -80,7 +80,7 @@ } ], "methodology": "Search capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "300-700ms average", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 90, @@ -127,7 +127,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "private_endpoints": { "score": 88, @@ -141,7 +141,7 @@ } ], "methodology": "Network security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 87, @@ -155,7 +155,7 @@ } ], "methodology": "Audit capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance": { "score": 91, @@ -169,7 +169,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 90, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "hipaa_compliance": { "score": 89, @@ -216,7 +216,7 @@ } ], "methodology": "Healthcare compliance review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_masking": { "score": 86, @@ -230,7 +230,7 @@ } ], "methodology": "PII protection testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_location": { "score": 89, @@ -244,7 +244,7 @@ } ], "methodology": "Data residency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "analytics_dashboard": { "score": 85, @@ -277,7 +277,7 @@ } ], "methodology": "Analytics capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "testing_tools": { "score": 82, @@ -291,7 +291,7 @@ } ], "methodology": "Testing features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "explainability": { "score": 80, @@ -305,7 +305,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 90, @@ -335,10 +335,16 @@ "url": "https://www.ibm.com/cloud/watson-assistant", "date": "2024-10-01", "value": "Enterprise-grade scalability on IBM Cloud" + }, + { + "source": "IBM watsonx Orchestrate", + "url": "https://www.ibm.com/products/watsonx-orchestrate", + "date": "2026-07-09", + "value": "Assistant capabilities now delivered within watsonx Orchestrate; next-generation multi-agent orchestration announced at IBM Think 2026 (May 2026)" } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 83, @@ -349,10 +355,16 @@ "url": "https://www.ibm.com/cloud/watson-assistant/pricing", "date": "2024-10-01", "value": "Free tier (1000 MAU), Plus $140/month, Enterprise custom pricing" + }, + { + "source": "IBM watsonx Orchestrate pricing", + "url": "https://www.ibm.com/products/watsonx-orchestrate/pricing", + "date": "2026-07-09", + "value": "watsonx Assistant pricing page now redirects to watsonx Orchestrate pricing; Orchestrate Essentials starts around $500/month, Standard/Premium custom-priced; legacy Assistant plans (Lite free, Plus from $140/month) remain for existing customers" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 86, @@ -366,7 +378,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "channel_integrations": { "score": 88, @@ -380,7 +392,7 @@ } ], "methodology": "Integration options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "industry_solutions": { "score": 85, @@ -394,7 +406,7 @@ } ], "methodology": "Industry solutions assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -408,6 +420,7 @@ "Comprehensive analytics and conversation insights" ], "limitations": [ + "Product consolidated into watsonx Orchestrate: standalone watsonx Assistant is no longer IBM's lead offering, its pricing page redirects to Orchestrate, and new implementations should start there; repeated rebranding (Watson Assistant -> watsonx Assistant -> watsonx Orchestrate) makes older documentation and links stale", "Higher cost compared to modern cloud alternatives", "IBM Cloud platform lock-in and complexity", "Steeper learning curve for dialog tree design", @@ -436,7 +449,7 @@ "Watson services", "Custom integrations" ], - "pricing_model": "Free (1000 MAU), Plus ($140/month), Enterprise (custom)", + "pricing_model": "Sold via watsonx Orchestrate: Essentials from ~$500/month, Standard/Premium custom; legacy Assistant plans (Lite free, Plus from $140/month) for existing customers", "supported_languages": "13 languages", "industry_packs": [ "Banking", @@ -450,7 +463,10 @@ "On-premises option" ], "cloud_pak": "Available in Cloud Pak for Data", - "pricing": "Free plan available, Paid plans from $140/month, Enterprise custom pricing" + "pricing": "New purchases via watsonx Orchestrate (Essentials from ~$500/month, higher tiers custom); legacy watsonx Assistant plans (Lite free, Plus from $140/month) remain for existing customers", + "successor_products": [ + "IBM watsonx Orchestrate" + ] }, "use_case_ratings": { "customer-support": { @@ -502,6 +518,7 @@ ], "tags": [ "ibm", - "enterprise" + "enterprise", + "rebranded" ] } diff --git a/data/agents/kore-ai.json b/data/agents/kore-ai.json index 2de6aa1..91d5a1e 100644 --- a/data/agents/kore-ai.json +++ b/data/agents/kore-ai.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Kore.ai", "provider": "Kore.ai Inc.", - "version": "11.0", - "last_evaluated": "2025-01-14", + "version": "Agent Platform - Artemis edition (2026); XO Platform v11", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Enterprise-grade agentic AI platform for designing, deploying, managing, and scaling AI agents across business operations. Offers no-code builders, pre-built industry solutions, and comprehensive orchestration capabilities for complex enterprise workflows.", + "description": "Enterprise agentic AI platform for designing, deploying, managing, and scaling AI agents across business operations. In May 2026 Kore.ai launched the Artemis edition of its Agent Platform, built around the Agent Blueprint Language (ABL), a compiled, declarative YAML-based language for defining, validating, and governing agents and multi-agent systems, initially on Microsoft Azure. Secured a strategic growth investment led by AllianceBernstein in January 2026 (total funding ~$620M).", "website": "https://kore.ai/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Enterprise deployment analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "natural_language_understanding": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "NLU capability testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "multi_channel_support": { "score": 92, @@ -52,7 +52,7 @@ } ], "methodology": "Channel integration testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "workflow_orchestration": { "score": 88, @@ -63,10 +63,16 @@ "url": "https://developer.kore.ai/", "date": "2025-01-10", "value": "Sophisticated workflow orchestration with dialog management" + }, + { + "source": "Kore.ai Artemis Agent Platform launch", + "url": "https://www.kore.ai/news", + "date": "2026-07-09", + "value": "Artemis edition of the Kore.ai Agent Platform launched May 2026 with Agent Blueprint Language (ABL) for defining, validating, and governing agents, workflows, and multi-agent systems; initially available on Microsoft Azure" } ], "methodology": "Orchestration capability review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "scalability": { "score": 90, @@ -80,7 +86,7 @@ } ], "methodology": "Scalability assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -99,7 +105,7 @@ } ], "methodology": "Security certification review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_encryption": { "score": 90, @@ -113,7 +119,7 @@ } ], "methodology": "Encryption review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "access_control": { "score": 92, @@ -127,7 +133,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 90, @@ -141,7 +147,7 @@ } ], "methodology": "Audit capability review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -155,7 +161,7 @@ } ], "methodology": "PII handling assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -174,7 +180,7 @@ } ], "methodology": "GDPR compliance review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "hipaa_compliance": { "score": 88, @@ -188,7 +194,7 @@ } ], "methodology": "HIPAA compliance review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_residency": { "score": 90, @@ -202,7 +208,7 @@ } ], "methodology": "Data residency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "score": 85, @@ -216,7 +222,7 @@ } ], "methodology": "Retention policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -235,7 +241,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "conversation_history": { "score": 90, @@ -249,7 +255,7 @@ } ], "methodology": "Logging capability assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "model_transparency": { "score": 78, @@ -263,7 +269,7 @@ } ], "methodology": "Model transparency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "performance_metrics": { "score": 85, @@ -277,7 +283,7 @@ } ], "methodology": "Analytics review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -296,7 +302,7 @@ } ], "methodology": "Deployment complexity assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "integration_ecosystem": { "score": 90, @@ -310,7 +316,7 @@ } ], "methodology": "Integration ecosystem review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -324,7 +330,7 @@ } ], "methodology": "Support quality assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "customization": { "score": 85, @@ -338,7 +344,7 @@ } ], "methodology": "Customization capability review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "industry_solutions": { "score": 88, @@ -352,7 +358,7 @@ } ], "methodology": "Industry solution review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } @@ -430,7 +436,14 @@ "certifications": ["SOC 2 Type II", "ISO 27001", "HIPAA", "GDPR"], "founded": "2014", "headquarters": "Orlando, FL", - "customers": "Fortune 500 companies" + "customers": "Fortune 500 companies", + "funding": "~$620.9M total (as of April 2026), including $150M growth round (2024) and a strategic growth investment led by AllianceBernstein Private Credit Investors (January 2026)", + "products": [ + "Agent Platform (Artemis edition, May 2026)", + "XO Platform v11", + "AI for Service", + "AI for Work" + ] }, "tags": ["enterprise", "conversational-ai", "customer-support", "no-code"] } diff --git a/data/agents/langflow.json b/data/agents/langflow.json index 8957d1d..d63c144 100644 --- a/data/agents/langflow.json +++ b/data/agents/langflow.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Langflow", "provider": "IBM (formerly DataStax/Logspace)", - "version": "1.5.1+", - "last_evaluated": "2026-06-10", + "version": "1.10.1", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Visual, drag-and-drop interface for building LangChain-based AI apps and agents; now owned by IBM via the Feb 2025 DataStax acquisition. SECURITY: serious CVE history including CVE-2025-3248 (CVSS 9.8 unauthenticated RCE, fixed in 1.3.0, on CISA KEV, exploited by the Flodrix botnet) and CVE-2025-34291 (account takeover/RCE). Patch promptly and harden deployments.", + "description": "Visual builder for LangChain-based AI apps and agents; owned by IBM via the Feb 2025 DataStax acquisition. SECURITY: recurring CVEs - CVE-2025-3248 (CVSS 9.8 unauthenticated RCE, fixed in 1.3.0, CISA KEV, Flodrix botnet), CVE-2025-34291 (account takeover/RCE), CVE-2026-33017 (second 9.8 unauthenticated RCE, exploited in 2026 for cryptomining, fixed in 1.9.0), CVE-2026-55255 (IDOR, fixed in 1.9.2), CVE-2026-5027 (path traversal). Upgrade to 1.10.1+; never expose unauthenticated.", "website": "https://www.langflow.org/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Workflow execution testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "langchain_compatibility": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Compatibility testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rapid_prototyping": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Development speed assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_support": { "score": 84, @@ -66,7 +66,7 @@ } ], "methodology": "LLM integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 72, @@ -80,7 +80,7 @@ } ], "methodology": "Error recovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Varies by flow (1-10s)", @@ -94,12 +94,12 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 67, + "overall_score": 64, "criteria": { "api_key_management": { "score": 58, @@ -119,7 +119,7 @@ } ], "methodology": "Security configuration review", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Score reduced due to CVE-2025-34291 account takeover/RCE" }, "self_hosting": { @@ -137,11 +137,17 @@ "url": "https://nvd.nist.gov/vuln/detail/CVE-2025-3248", "date": "2026-06-10", "value": "Internet-exposed self-hosted Langflow instances were actively exploited via CVE-2025-3248 (added to CISA KEV 2025-05-05; exploited by the Flodrix botnet)" + }, + { + "source": "Orca Security / Trend Micro - CVE-2026-33017 exploitation", + "url": "https://nvd.nist.gov/vuln/detail/cve-2026-33017", + "date": "2026-07-09", + "value": "CVE-2026-33017 (CVSS 9.8 unauthenticated RCE via /api/v1/build_public_tmp) actively exploited in spring 2026 against exposed instances to deploy XMRig cryptominers; affects <=1.8.2, fixed in 1.9.0" } ], "methodology": "Deployment security assessment", - "last_verified": "2026-06-10", - "notes": "Score reduced: active in-the-wild exploitation of exposed self-hosted deployments" + "last_verified": "2026-07-09", + "notes": "Score reduced: repeated active in-the-wild exploitation of exposed self-hosted deployments (CVE-2025-3248 in 2025, CVE-2026-33017 in 2026)" }, "data_privacy": { "score": 78, @@ -155,7 +161,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 90, @@ -163,16 +169,16 @@ "evidence": [ { "source": "GitHub", - "url": "https://github.com/logspace-ai/langflow", - "date": "2024-10-20", - "value": "MIT license, 30k+ stars, open source community" + "url": "https://github.com/langflow-ai/langflow", + "date": "2026-07-09", + "value": "MIT license, 151k+ stars, open source community" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "authentication": { - "score": 40, + "score": 36, "confidence": "high", "evidence": [ { @@ -186,11 +192,23 @@ "url": "https://nvd.nist.gov/vuln/detail/CVE-2025-3248", "date": "2026-06-10", "value": "CVE-2025-3248: CVSS 9.8 unauthenticated remote code execution via /api/v1/validate/code; fixed in 1.3.0; added to CISA KEV 2025-05-05 and exploited by the Flodrix botnet" + }, + { + "source": "CVE-2026-33017", + "url": "https://nvd.nist.gov/vuln/detail/cve-2026-33017", + "date": "2026-07-09", + "value": "CVE-2026-33017: second CVSS 9.8 unauthenticated RCE (public flow-building API endpoint), actively exploited in spring 2026 within 20 hours of advisory publication; fixed in 1.9.0 (1.8.2 widely reported as patched but still vulnerable)" + }, + { + "source": "GHSA-qrpv-q767-xqq2 (CVE-2026-55255)", + "url": "https://github.com/advisories/GHSA-qrpv-q767-xqq2", + "date": "2026-07-09", + "value": "CVE-2026-55255: IDOR in /api/v1/responses lets authenticated attackers execute other users' flows; fixed in 1.9.2" } ], "methodology": "Authentication assessment", - "last_verified": "2026-06-10", - "notes": "Score reduced due to CVE-2025-3248 unauthenticated RCE with confirmed in-the-wild exploitation" + "last_verified": "2026-07-09", + "notes": "Score reduced further 2026-07-09: recurring pattern of unauthenticated RCEs with confirmed in-the-wild exploitation (CVE-2025-3248, CVE-2026-33017) plus IDOR CVE-2026-55255; all patched as of 1.9.2, current release 1.10.1" } } }, @@ -209,7 +227,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 76, @@ -223,7 +241,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 88, @@ -237,7 +255,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 72, @@ -251,7 +269,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "flow_export": { "score": 82, @@ -265,7 +283,7 @@ } ], "methodology": "Data portability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -284,7 +302,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "visual_debugging": { "score": 88, @@ -298,7 +316,7 @@ } ], "methodology": "Debugging tools assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 92, @@ -306,13 +324,13 @@ "evidence": [ { "source": "GitHub", - "url": "https://github.com/logspace-ai/langflow", - "date": "2024-10-20", - "value": "MIT license, 30k+ stars, very active community" + "url": "https://github.com/langflow-ai/langflow", + "date": "2026-07-09", + "value": "MIT license, 151k+ stars, very active community (repo active as of July 2026)" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 83, @@ -326,7 +344,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -345,7 +363,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 70, @@ -359,7 +377,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 90, @@ -373,7 +391,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 68, @@ -387,7 +405,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 72, @@ -401,7 +419,7 @@ } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "template_library": { "score": 86, @@ -415,14 +433,14 @@ } ], "methodology": "Template availability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } }, "strengths": [ "Intuitive visual drag-and-drop interface for LLM workflows", - "Open source (MIT) with very active community (30k+ stars)", + "Open source (MIT) with very active community (151k+ stars)", "Built on LangChain ecosystem with access to all components", "Excellent for rapid prototyping and experimentation", "Low-code approach makes AI accessible to non-developers", @@ -435,7 +453,7 @@ "Security features less mature than enterprise platforms", "Limited control compared to code-based implementations", "Debugging complex flows can be challenging despite visual interface", - "Serious CVE history: CVE-2025-3248 (unauthenticated RCE, CISA KEV, Flodrix botnet) and CVE-2025-34291 (account takeover/RCE); upgrade to 1.3.0+ and never expose unauthenticated instances" + "Recurring critical CVE pattern: CVE-2025-3248 (unauthenticated RCE, CISA KEV, Flodrix botnet), CVE-2025-34291 (account takeover/RCE), CVE-2026-33017 (second unauthenticated RCE, actively exploited spring 2026 for cryptomining), CVE-2026-55255 (IDOR) and CVE-2026-5027 (path traversal); upgrade to 1.10.1+ and never expose unauthenticated instances" ], "metadata": { "license": "MIT", @@ -456,12 +474,12 @@ "LangChain tools", "Custom Python components" ], - "pricing_model": "Free open source (DataStax offers managed enterprise version)", - "github_stars": "130000+", + "pricing_model": "Free open source (IBM/DataStax offers managed enterprise version)", + "github_stars": "151000+", "first_release": "2023", "ui_framework": "React-based visual interface", "langchain_version": "Compatible with LangChain ecosystem", - "version": "1.5.1+", + "version": "1.10.1 (June 29, 2026)", "pricing": "Free (open source), Cloud from $0/hour (usage-based)" }, "use_case_ratings": { diff --git a/data/agents/langgraph-agent.json b/data/agents/langgraph-agent.json index 5c8f5fd..c67c7c5 100644 --- a/data/agents/langgraph-agent.json +++ b/data/agents/langgraph-agent.json @@ -3,8 +3,8 @@ "type": "agent", "name": "LangGraph Agent", "provider": "LangChain", - "version": "1.0", - "last_evaluated": "2026-06-10", + "version": "1.x", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "LangChain's graph-based agent framework for building stateful, multi-actor applications with cycles and controllable execution flow. Reached 1.0 GA on 2025-10-22, adding durable execution and middleware. Enables complex, cyclic agent workflows with human-in-the-loop capabilities and production-grade persistence.", "website": "https://langchain-ai.github.io/langgraph/", @@ -24,7 +24,7 @@ } ], "methodology": "Based on underlying model performance and framework overhead", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing with graph execution", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 87, @@ -66,7 +66,7 @@ } ], "methodology": "Memory persistence testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Variable (2-8s typical)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 78, @@ -127,7 +127,7 @@ } ], "methodology": "Access control capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 80, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 85, @@ -152,10 +152,17 @@ "url": "https://langchain-ai.github.io/langgraph/how-tos/persistence/", "date": "2024-10-01", "value": "Thread-based isolation when using proper checkpointing" + }, + { + "source": "Check Point Research - Exploiting LangGraph's Checkpointer", + "url": "https://research.checkpoint.com/2026/from-sqli-to-rce-exploiting-langgraphs-checkpointer/", + "date": "2026-03-29", + "value": "Checkpointer CVEs disclosed: CVE-2025-64439 (JsonPlusSerializer json-mode RCE, fixed in langgraph-checkpoint 3.0), CVE-2025-67644 (SQLite checkpointer SQLi), CVE-2026-28277 (unsafe msgpack deserialization to RCE), CVE-2026-27022 (Redis checkpointer injection); patched in langgraph 1.0.10+, langgraph-checkpoint-sqlite 3.0.1+, langgraph-checkpoint-redis 1.0.2+" } ], - "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "methodology": "Data architecture review plus disclosed vulnerability analysis", + "last_verified": "2026-07-09", + "notes": "Score held at 85: the March 2026 checkpointer vulnerability cluster was patched promptly and current 1.2.x releases are well past the fixed versions; self-hosters on SQLite/Redis checkpointers must keep dependencies current" }, "open_source_transparency": { "score": 95, @@ -169,7 +176,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +195,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 85, @@ -202,7 +209,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 75, @@ -216,7 +223,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 95, @@ -230,7 +237,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +256,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 90, @@ -263,7 +270,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 82, @@ -277,7 +284,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 95, @@ -291,7 +298,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -310,7 +317,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 85, @@ -324,7 +331,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 90, @@ -338,7 +345,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 88, @@ -352,7 +359,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 82, @@ -369,10 +376,16 @@ "url": "https://blog.langchain.com/langchain-langgraph-1dot0/", "date": "2026-06-10", "value": "LangGraph 1.0 GA released 2025-10-22 with durable execution and middleware; large, mature ecosystem" + }, + { + "source": "langgraph on PyPI", + "url": "https://pypi.org/project/langgraph/", + "date": "2026-07-09", + "value": "Steady 1.x cadence: latest release 1.2.8 (2026-07-06); repo at ~36,900 stars with daily activity; 1.x promises no breaking changes until 2.0" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Score raised: 1.0 GA milestone and ecosystem maturity" } } @@ -390,10 +403,9 @@ "Steeper learning curve compared to simpler agent frameworks", "Requires more code for simple use cases", "Security and sandboxing must be implemented by developer", - "Relatively newer framework with evolving APIs", "Performance overhead from graph execution layer", - "Limited managed service options (LangGraph Cloud in beta)", - "Status (2026-06): 1.0 GA since 2025-10-22 (durable execution, middleware); earlier evolving-API concerns are largely resolved" + "March 2026 checkpointer CVEs (SQLi and deserialization to RCE) require self-hosters to stay on langgraph 1.0.10+/current checkpoint packages", + "Status (2026-07): 1.x stable (1.2.8 as of 2026-07-06); 1.0 GA since 2025-10-22 (durable execution, middleware); earlier evolving-API concerns are resolved and managed deployment is GA via LangGraph Platform" ], "metadata": { "license": "MIT", @@ -407,17 +419,20 @@ "Python", "JavaScript/TypeScript" ], - "deployment_type": "Self-hosted or LangGraph Cloud", + "deployment_type": "Self-hosted or LangGraph Platform (managed, GA)", "tool_support": [ "LangChain tools", "Custom tools", "Function calling" ], - "github_stars": "11700+", + "github_stars": "36800+", "monthly_downloads": "7M+ on PyPI", "first_release": "2024", "pricing": "Free (MIT license) - Cloud deployment via LangSmith Plus/Enterprise plans", - "ga_version": "1.0 (October 2025)" + "pricing_last_verified": "2026-07-09", + "ga_version": "1.0 (October 2025)", + "current_version": "1.2.8 (2026-07-06)", + "security_advisories": "CVE-2025-64439, CVE-2025-67644, CVE-2026-28277, CVE-2026-27022 (checkpointer SQLi/deserialization, disclosed March 2026; fixed in langgraph 1.0.10+, langgraph-checkpoint-sqlite 3.0.1+, langgraph-checkpoint-redis 1.0.2+)" }, "use_case_ratings": { "customer-support": { diff --git a/data/agents/llamaindex-agent.json b/data/agents/llamaindex-agent.json index 7743e39..aaa8a43 100644 --- a/data/agents/llamaindex-agent.json +++ b/data/agents/llamaindex-agent.json @@ -3,10 +3,10 @@ "type": "agent", "name": "LlamaIndex Agent", "provider": "LlamaIndex", - "version": "0.11.x", - "last_evaluated": "2025-11-09", + "version": "0.14.x", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Data framework optimized for building LLM applications with advanced RAG (Retrieval-Augmented Generation) capabilities. Agents can reason over complex data sources using sophisticated query engines and retrieval strategies.", + "description": "Data framework optimized for building LLM applications with advanced RAG (Retrieval-Augmented Generation) capabilities. Agents can reason over complex data sources using sophisticated query engines and retrieval strategies; the event-driven Workflows engine (llama-index-workflows 2.x) has matured agent orchestration well beyond the earlier ReAct-only story. Managed cloud offerings (LlamaCloud parse/extract/index, LlamaAgents) are available alongside the MIT-licensed framework.", "website": "https://docs.llamaindex.ai/en/stable/module_guides/deploying/agents/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on RAG performance benchmarks", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 84, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 78, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rag_performance": { "score": 92, @@ -94,7 +94,7 @@ } ], "methodology": "RAG benchmark testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 75, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 78, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 82, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 94, @@ -166,10 +166,16 @@ "url": "https://github.com/run-llama/llama_index", "date": "2024-10-20", "value": "MIT licensed, 35k+ stars, very active development" + }, + { + "source": "GitHub Repository", + "url": "https://github.com/run-llama/llama_index", + "date": "2026-07-09", + "value": "Now ~50,700 stars, still very active (pushed 2026-07-08); a series of 2025 CVEs in reader/CLI components (CVE-2025-1753 CLI command injection fixed in 0.12.21, CVE-2025-5302 JSONReader DoS, CVE-2025-7647/CVE-2025-7707 insecure temp/cache file handling) were disclosed and patched via the public advisory process; current 0.14.x releases are past all published fixes" } ], - "methodology": "Source code review", - "last_verified": "2025-11-09" + "methodology": "Source code review plus advisory history analysis", + "last_verified": "2026-07-09" } } }, @@ -188,7 +194,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 80, @@ -202,7 +208,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 75, @@ -216,7 +222,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 93, @@ -230,7 +236,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +255,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 83, @@ -263,7 +269,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 85, @@ -277,7 +283,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 94, @@ -291,7 +297,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_activity": { "score": 90, @@ -305,7 +311,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +330,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 83, @@ -338,7 +344,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -352,7 +358,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 80, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rag_optimization": { "score": 91, @@ -380,7 +386,7 @@ } ], "methodology": "RAG capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -394,9 +400,10 @@ "Excellent for knowledge-intensive applications" ], "limitations": [ - "Agent capabilities less mature than core RAG features", + "Agent orchestration improved substantially with the Workflows engine, but the framework remains RAG-first compared to dedicated agent frameworks", "Requires understanding of RAG concepts for optimal use", "Security and sandboxing must be implemented separately", + "Repeated 2025 CVEs in reader/CLI components (command injection, DoS, insecure temp files) mean data connectors should be treated as an attack surface and dependencies kept current", "Can have high latency with complex retrieval strategies", "Embedding costs can accumulate with large datasets", "Less suitable for tasks not requiring data retrieval" @@ -420,8 +427,12 @@ "Function tools", "Custom tools" ], - "github_stars": "35000+", - "first_release": "2022" + "github_stars": "50700+", + "first_release": "2022", + "current_version": "0.14.23 (2026-06-24); llama-index-workflows 2.22.2 (2026-06-30)", + "pricing": "Framework free (MIT) - Costs from LLM API and embedding services. LlamaCloud managed platform: credit-based, Free tier 10,000 credits/month, Starter $50/month, Pro $500/month, custom Enterprise", + "pricing_last_verified": "2026-07-09", + "security_advisories": "2025 CVEs patched: CVE-2025-1753 (CLI command injection, fixed 0.12.21), CVE-2025-1752/CVE-2025-5302 (reader DoS), CVE-2025-7647/CVE-2025-7707 (insecure temp/cache files); current 0.14.x is past all published fixes" }, "use_case_ratings": { "customer-support": { diff --git a/data/agents/make-ai.json b/data/agents/make-ai.json index 40fe8a2..efa47fc 100644 --- a/data/agents/make-ai.json +++ b/data/agents/make-ai.json @@ -4,9 +4,9 @@ "name": "Make AI", "provider": "Make (Integromat)", "version": "Current", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Visual automation platform with AI capabilities for building no-code intelligent workflows. Integrates AI models with 1500+ app connections for creating automated business processes and intelligent agents.", + "description": "Visual automation platform with AI capabilities for building no-code intelligent workflows. Integrates AI models with 3,000+ app connections for creating automated business processes, and offers Make AI Agents (available on all paid plans) that can reason, decide next steps, and trigger workflows.", "website": "https://www.make.com/en/ai", "trust_vector": { "performance_reliability": { @@ -24,21 +24,21 @@ } ], "methodology": "Workflow reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ai_module_performance": { "score": 87, "confidence": "high", "evidence": [ { - "source": "AI Modules", - "url": "https://www.make.com/en/integrations/openai", - "date": "2024-10-15", - "value": "OpenAI, Anthropic, and custom AI model integrations" + "source": "AI Modules / Make AI Agents", + "url": "https://www.make.com/en/ai-agents", + "date": "2026-07-09", + "value": "OpenAI, Anthropic, and custom AI model integrations; Make AI Agents available on all paid plans with access to 3,000+ integrations" } ], "methodology": "AI integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "integration_reliability": { "score": 92, @@ -47,12 +47,12 @@ { "source": "Integrations", "url": "https://www.make.com/en/integrations", - "date": "2024-10-01", - "value": "1500+ pre-built app integrations with high reliability" + "date": "2026-07-09", + "value": "3,000+ pre-built app integrations with high reliability" } ], "methodology": "Integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_processing": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Data processing testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 84, @@ -80,7 +80,7 @@ } ], "methodology": "Error recovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "200ms-5s average", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "authentication": { "score": 90, @@ -127,7 +127,7 @@ } ], "methodology": "Authentication testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_vault": { "score": 89, @@ -141,7 +141,7 @@ } ], "methodology": "Credential security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance": { "score": 91, @@ -155,7 +155,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "webhook_security": { "score": 82, @@ -169,7 +169,7 @@ } ], "methodology": "Webhook security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 90, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_residency": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Data residency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_processing": { "score": 84, @@ -230,7 +230,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logs": { "score": 87, @@ -244,7 +244,7 @@ } ], "methodology": "Audit capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "visual_builder": { "score": 93, @@ -277,7 +277,7 @@ } ], "methodology": "UI/UX assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_history": { "score": 86, @@ -291,7 +291,7 @@ } ], "methodology": "Traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "template_marketplace": { "score": 84, @@ -305,7 +305,7 @@ } ], "methodology": "Template availability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "support_quality": { "score": 76, @@ -319,7 +319,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 90, @@ -352,7 +352,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 81, @@ -361,12 +361,12 @@ { "source": "Pricing", "url": "https://www.make.com/en/pricing", - "date": "2024-10-01", - "value": "Operation-based pricing, free tier + paid plans from $9/month" + "date": "2026-07-09", + "value": "Operation-based pricing: Free (1,000 ops/month), Core from $9/month (annual, 10,000 ops; $10.59 monthly), Pro from $16/month, Teams from $29/month, Enterprise custom; unused operations roll over one month on paid plans (2026 feature)" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 87, @@ -380,7 +380,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "integration_ecosystem": { "score": 96, @@ -389,12 +389,12 @@ { "source": "Apps", "url": "https://www.make.com/en/integrations", - "date": "2024-10-01", - "value": "1500+ app integrations, largest in the market" + "date": "2026-07-09", + "value": "3,000+ app integrations, one of the largest ecosystems (behind Zapier's 8,000+)" } ], "methodology": "Integration ecosystem assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "uptime_sla": { "score": 89, @@ -408,14 +408,15 @@ } ], "methodology": "SLA review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } }, "strengths": [ "No-code visual builder with exceptional user experience", - "1500+ pre-built integrations, largest app ecosystem", + "3,000+ pre-built integrations, one of the largest app ecosystems", + "Make AI Agents included on all paid plans for adaptive, multi-step automation", "Enterprise-grade security and compliance (SOC 2, ISO 27001, GDPR)", "Excellent documentation and template marketplace", "99.9% uptime SLA with reliable infrastructure", @@ -434,7 +435,7 @@ "supported_models": [ "OpenAI", "Anthropic Claude", - "Google PaLM", + "Google Gemini", "Custom AI APIs" ], "programming_languages": [ @@ -443,16 +444,17 @@ ], "deployment_type": "Managed cloud service", "tool_support": [ - "1500+ app integrations", + "3,000+ app integrations", "HTTP requests", - "Custom functions" + "Custom functions", + "Make AI Agents" ], - "pricing_model": "Free tier + paid plans ($9-$299/month)", + "pricing_model": "Free tier + paid plans (Core from $9/month annual)", "operations_included": "1,000 to unlimited per month", "uptime_sla": "99.9% (Enterprise)", "data_centers": "US, EU", "founded": "2012 (as Integromat, rebranded to Make 2021)", - "pricing": "Free: 1,000 operations/month; Core: $9-10.59/month for 10k operations; Pro: $18.82/month; Teams: $29-34.12/month; Enterprise: Custom" + "pricing": "Free: 1,000 operations/month; Core: $9/month (annual) or $10.59/month for 10k operations; Pro: from $16/month; Teams: from $29/month; Enterprise: Custom; unused operations roll over one month on paid plans (verified 2026-07-09)" }, "use_case_ratings": { "customer-support": { diff --git a/data/agents/manus.json b/data/agents/manus.json index 221bd3a..3ed98d0 100644 --- a/data/agents/manus.json +++ b/data/agents/manus.json @@ -2,11 +2,11 @@ "id": "manus", "type": "agent", "name": "Manus", - "provider": "Meta (formerly Butterfly Effect)", + "provider": "Butterfly Effect / Meta (acquisition being unwound as of June 2026)", "version": "1.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "General-purpose autonomous agent that executes end-to-end tasks in a cloud VM equipped with a browser, shell, and file tools. Launched virally in March 2025 by Butterfly Effect and acquired by Meta in a deal that closed in late December 2025 for over $2B.", + "description": "General-purpose autonomous agent that executes end-to-end tasks in a cloud VM equipped with a browser, shell, and file tools. Launched virally in March 2025 by Butterfly Effect and acquired by Meta for over $2B in late December 2025 — but Chinese regulators ordered the deal reversed in April 2026, and by June 2026 Meta had completed an operational split (halting data sharing) while it dismantles the acquisition; Manus co-founders are reportedly in talks to buy the company back.", "website": "https://manus.im/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Assessment of end-to-end task completion based on published benchmark claims and independent user testing reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 80, @@ -38,7 +38,7 @@ } ], "methodology": "Review of integrated tool stack reliability across browsing, shell, and file operations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Evaluation of autonomous task decomposition and long-horizon asynchronous execution", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 75, @@ -66,7 +66,7 @@ } ], "methodology": "Review of session workspace persistence and cross-task preference memory", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 72, @@ -80,7 +80,7 @@ } ], "methodology": "Assessment of retry behavior and failure modes from independent reviews", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 70, @@ -94,7 +94,7 @@ } ], "methodology": "Review of internal multi-agent architecture versus user-controllable collaboration", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of cloud VM isolation model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 65, @@ -127,7 +127,7 @@ } ], "methodology": "Review of credential handling and account-level controls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 58, @@ -141,21 +141,21 @@ } ], "methodology": "Threat surface analysis of autonomous browsing with limited disclosed defenses", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 66, "confidence": "low", "evidence": [ { - "source": "CNBC - Meta acquires Manus", - "url": "https://www.cnbc.com/2025/12/30/meta-acquires-singapore-ai-agent-firm-manus-china-butterfly-effect-monicai.html", - "date": "2025-12-30", - "value": "Per-session VM isolation exists, but data residency and processing arrangements are in transition following Meta's acquisition of the Singapore-based, China-originated company" + "source": "Bloomberg - Meta severs Manus data access", + "url": "https://www.bloomberg.com/news/articles/2026-06-11/meta-severs-manus-data-access-after-china-orders-buyout-unwound", + "date": "2026-06-11", + "value": "Per-session VM isolation exists, but data residency and processing arrangements are in flux again: after Beijing ordered the buyout unwound, Meta completed an operational split from Manus and halted data sharing between the companies in June 2026" } ], - "methodology": "Data architecture review accounting for ownership and jurisdiction transition", - "last_verified": "2026-06-10" + "methodology": "Data architecture review accounting for ownership and jurisdiction transition, updated for the June 2026 unwind and data-sharing cutoff", + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 28, @@ -169,7 +169,7 @@ } ], "methodology": "Source availability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -184,25 +184,25 @@ "source": "Manus Privacy Policy", "url": "https://manus.im/privacy", "date": "2026-05-01", - "value": "Session data, files, and browsing artifacts retained in Manus cloud; retention terms are consumer-grade and being restated under Meta ownership" + "value": "Session data, files, and browsing artifacts retained in Manus cloud; retention terms are consumer-grade and unsettled while the Meta acquisition is unwound" } ], "methodology": "Review of published retention practices during ownership transition", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 62, "confidence": "low", "evidence": [ { - "source": "CNBC - Meta acquires Manus", - "url": "https://www.cnbc.com/2025/12/30/meta-acquires-singapore-ai-agent-firm-manus-china-butterfly-effect-monicai.html", - "date": "2025-12-30", - "value": "Meta ownership brings established compliance infrastructure, but policies and processing locations for Manus are still being integrated post-acquisition" + "source": "CNBC - Meta begins dismantling Manus deal", + "url": "https://www.cnbc.com/2026/06/12/meta-reportedly-begins-dismantling-2-billion-manus-deal-on-beijings-orders.html", + "date": "2026-06-12", + "value": "Compliance ownership is unresolved: Chinese regulators ordered the $2B deal reversed (April 2026) and Meta is dismantling the acquisition, so the Meta compliance umbrella no longer reliably applies to Manus" } ], - "methodology": "Compliance posture assessment during corporate integration", - "last_verified": "2026-06-10" + "methodology": "Compliance posture assessment during corporate separation", + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 55, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis of third-party model routing before and after acquisition", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 25, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 80, @@ -263,7 +263,7 @@ } ], "methodology": "Review of live session visibility and replay features", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 72, @@ -277,7 +277,7 @@ } ], "methodology": "Assessment of plan visibility and progress narration", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 30, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 72, @@ -305,12 +305,12 @@ } ], "methodology": "Community engagement analysis of user base and public activity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 69, + "overall_score": 68, "criteria": { "ease_of_integration": { "score": 78, @@ -324,7 +324,7 @@ } ], "methodology": "Onboarding friction assessment for non-technical users", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 74, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability assessment of concurrent task limits across tiers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 62, @@ -352,7 +352,7 @@ } ], "methodology": "Pricing model analysis; variable credit burn per task reduces predictability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 65, @@ -366,21 +366,21 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { - "score": 68, + "score": 60, "confidence": "medium", "evidence": [ { - "source": "CNBC - Meta acquires Manus", - "url": "https://www.cnbc.com/2025/12/30/meta-acquires-singapore-ai-agent-firm-manus-china-butterfly-effect-monicai.html", - "date": "2025-12-30", - "value": "Meta acquisition (>$2B, closed ~2025-12-30) secures resources and continuity, but product, policy, and roadmap are mid-integration" + "source": "CNBC - Meta begins dismantling Manus deal", + "url": "https://www.cnbc.com/2026/06/12/meta-reportedly-begins-dismantling-2-billion-manus-deal-on-beijings-orders.html", + "date": "2026-06-12", + "value": "Meta began dismantling the $2B acquisition after Beijing ordered it reversed under China's foreign investment security review; operational split and data-sharing halt completed June 2026; co-founders reportedly seeking ~$1B to buy the company back, leaving ownership, resourcing, and roadmap unresolved" } ], - "methodology": "Vendor stability and product maturity assessment during acquisition integration", - "last_verified": "2026-06-10" + "methodology": "Vendor stability and product maturity assessment; score lowered 68 to 60 to reflect the forced unwind of the Meta acquisition and unresolved future ownership", + "last_verified": "2026-07-09" } } } @@ -418,11 +418,10 @@ "Transparent 'Manus's Computer' live view and replayable sessions show exactly what the agent did", "Asynchronous execution continues in the cloud after the user disconnects", "Zero-setup web and mobile experience accessible to non-technical users", - "Meta acquisition (closed ~2025-12-30, >$2B) provides long-term resourcing and infrastructure", "Free tier with 300 daily credits allows meaningful evaluation before paying" ], "limitations": [ - "Governance and jurisdiction transition (Singapore/China origins to Meta ownership) leaves data handling policies in flux", + "Ownership crisis: Beijing ordered Meta's $2B acquisition reversed (April 2026); Meta completed an operational split and halted data sharing in June 2026 while co-founders seek a buyback, leaving governance, data handling, and long-term resourcing unresolved", "Minimal published security documentation; prompt injection defenses for autonomous browsing are unclear", "Credit-based pricing with variable per-task burn makes costs hard to predict", "Cloud-only with no self-hosted option; sensitive data must enter Manus's VMs", @@ -434,7 +433,7 @@ "supported_models": [ "Anthropic Claude (historical backend)", "Alibaba Qwen (historical backend)", - "Meta model integration in progress post-acquisition" + "Model routing in flux amid the Meta acquisition unwind" ], "programming_languages": [ "Natural language interface; agent writes Python, JavaScript, and shell internally" @@ -449,7 +448,7 @@ ], "first_release": "2025-03-06 (viral invite launch); open signup May 2025", "pricing": "Free (300 daily credits); paid plans ~$20-40/mo up to $200/mo (credit-based)", - "company_milestones": "Built by Butterfly Effect (Singapore HQ, Chinese origins); acquired by Meta for >$2B, deal closed ~2025-12-30" + "company_milestones": "Built by Butterfly Effect (Singapore HQ, Chinese origins); acquired by Meta for >$2B (closed ~2025-12-30); Chinese regulators ordered the deal unwound April 2026; Meta completed operational split and halted data sharing June 2026; co-founders reportedly raising ~$1B for a buyback" }, "related_entities": [ "devin", diff --git a/data/agents/mastra.json b/data/agents/mastra.json index 6ed66f8..8bed148 100644 --- a/data/agents/mastra.json +++ b/data/agents/mastra.json @@ -4,7 +4,7 @@ "name": "Mastra", "provider": "Mastra AI (YC W25)", "version": "1.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "TypeScript-first AI agent framework from the Gatsby founders, combining agents, durable workflows, RAG, and evals in one toolkit. Provider-agnostic via Vercel AI SDK model routing, with a local dev playground and 1.0 stable release in January 2026.", "website": "https://mastra.ai/", @@ -24,7 +24,7 @@ } ], "methodology": "Task completion testing across agent and workflow primitives", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Tool invocation testing with typed schemas and MCP servers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 86, @@ -52,7 +52,7 @@ } ], "methodology": "Complex multi-step task testing using workflow engine", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 84, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation across threads and semantic recall", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Error injection testing on workflow retry and resume paths", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 82, @@ -94,7 +94,7 @@ } ], "methodology": "Multi-agent coordination testing via sub-agents and networks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of tool execution model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 70, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment of server auth options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 65, @@ -141,7 +141,7 @@ } ], "methodology": "Injection testing with and without guardrail processors configured", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 72, @@ -155,7 +155,7 @@ } ], "methodology": "Data isolation architecture review of storage and memory scoping", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 90, @@ -169,7 +169,7 @@ } ], "methodology": "Source code and license structure review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review of self-hosted deployment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 78, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment of deployment configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 76, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis across model provider configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 88, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 85, @@ -263,7 +263,7 @@ } ], "methodology": "Tracing and logging capabilities assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 78, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability assessment of run visualization features", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 88, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment of license split and code availability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 88, @@ -302,10 +302,16 @@ "url": "https://github.com/mastra-ai/mastra", "date": "2026-06-01", "value": "22k+ GitHub stars, 300k+ weekly npm downloads, active Discord and rapid release cadence" + }, + { + "source": "npm (mastra, @mastra/core)", + "url": "https://www.npmjs.com/package/@mastra/core", + "date": "2026-07-09", + "value": "Release cadence remains rapid post-1.0: mastra 1.15.1 and @mastra/core 1.50.0 published within a day of 2026-07-09; downloads grew to ~1.8M/month by Feb 2026" } ], "methodology": "Community engagement analysis of stars, downloads, and releases", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +330,7 @@ } ], "methodology": "Integration complexity assessment with scaffolded project testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 80, @@ -338,7 +344,7 @@ } ], "methodology": "Scalability assessment across deployment targets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 85, @@ -352,7 +358,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 84, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring and evaluation features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 84, @@ -380,7 +386,7 @@ } ], "methodology": "Production readiness assessment of API stability and release maturity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -461,6 +467,7 @@ ], "github_stars": "22000+", "first_release": "2024 (YC W25; 1.0 stable 2026-01-21)", + "latest_version": "mastra 1.15.1 / @mastra/core 1.50.0 (July 2026)", "pricing": "Free (Apache 2.0 core) - Costs from LLM APIs, hosting, and optional enterprise/cloud offerings", "node_requirement": "Node.js >=20", "adoption": "300k+ weekly npm downloads; built by the Gatsby founders" diff --git a/data/agents/memgpt.json b/data/agents/memgpt.json index 28c8e18..fe095af 100644 --- a/data/agents/memgpt.json +++ b/data/agents/memgpt.json @@ -4,9 +4,9 @@ "name": "Letta (formerly MemGPT)", "provider": "Letta Inc.", "version": "0.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "REBRANDED: MemGPT became Letta (Letta Inc., a UC Berkeley spinout) in September 2024; \"MemGPT\" now refers only to the underlying research technique. Letta is a memory-enhanced LLM agent platform enabling long-term context via virtual context management inspired by operating systems, overcoming context window limits for stateful agents.", + "description": "REBRANDED: MemGPT became Letta (Letta Inc., a UC Berkeley spinout) in September 2024; \"MemGPT\" now refers only to the research technique. Letta is a memory-enhanced LLM agent platform enabling long-term context via OS-inspired virtual context management. In March 2026 Letta announced a pivot to Letta Code, a client-side, model-agnostic agent harness with persistent memory, deprecating server-side memory tools, templates, identities, MCP integrations, and tool rules.", "website": "https://www.letta.com", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Memory system testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "long_context_handling": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Long context testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversation_continuity": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Continuity testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_integration": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "LLM integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_editing_memory": { "score": 76, @@ -80,7 +80,7 @@ } ], "methodology": "Self-editing capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "1-5s (memory overhead)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Isolation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_hosting": { "score": 87, @@ -127,7 +127,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_persistence": { "score": 80, @@ -141,7 +141,7 @@ } ], "methodology": "Data security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 90, @@ -155,7 +155,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_security": { "score": 68, @@ -169,7 +169,7 @@ } ], "methodology": "Memory security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 77, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 90, @@ -216,7 +216,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 72, @@ -230,7 +230,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_deletion": { "score": 76, @@ -244,7 +244,7 @@ } ], "methodology": "Data deletion assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "research_backed": { "score": 92, @@ -277,7 +277,7 @@ } ], "methodology": "Research foundation assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_visibility": { "score": 80, @@ -291,7 +291,7 @@ } ], "methodology": "Transparency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 88, @@ -305,7 +305,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 76, @@ -319,7 +319,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 74, @@ -352,7 +352,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -366,7 +366,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 70, @@ -380,7 +380,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 72, @@ -397,10 +397,17 @@ "url": "https://www.letta.com/blog/memgpt-and-letta", "date": "2026-06-10", "value": "MemGPT rebranded to Letta (Letta Inc., UC Berkeley spinout, Sept 2024); active commercial development continues under the Letta name" + }, + { + "source": "Letta's Next Phase (blog)", + "url": "https://www.letta.com/blog/our-next-phase/", + "date": "2026-07-09", + "value": "March 2026 pivot to Letta Code (open-source, client-side, model-agnostic agent harness with git-backed 'MemFS' memory); server-side memory tools, templates, identities, MCP integrations, and tool rules deprecated by mid-April 2026; letta server package latest 0.16.8 (May 2026), repo active (23.7k stars, pushed July 2026)" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09", + "notes": "Score held: development remains active, but the March 2026 architectural pivot and server-side feature deprecations add migration risk for teams built on the pre-2026 server APIs" }, "cli_interface": { "score": 82, @@ -414,7 +421,7 @@ } ], "methodology": "CLI capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -422,7 +429,7 @@ "strengths": [ "Breakthrough memory system enabling unlimited conversation context", "Research-backed approach (published paper) with novel virtual context", - "Open source (Apache 2.0) with active community (12k+ stars)", + "Open source (Apache 2.0) with active community (23k+ stars)", "Self-hosting option for complete memory control", "Supports multiple LLM providers and local models", "Enables truly stateful, personalized agent interactions" @@ -434,7 +441,7 @@ "Active development, some features still experimental", "Limited production tooling and monitoring features", "Memory pagination can introduce unpredictability", - "Rebranded to Letta (Sept 2024): older MemGPT docs, package names, and repo links are outdated" + "Rebranded to Letta (Sept 2024, older MemGPT docs/packages outdated); March 2026 pivot to client-side Letta Code harness deprecated server-side memory tools, templates, identities, MCP integrations, and tool rules by mid-April 2026, requiring migration for existing server-based deployments" ], "metadata": { "license": "Apache 2.0", @@ -447,18 +454,20 @@ "programming_languages": [ "Python" ], - "deployment_type": "Self-hosted (pip, Docker) or MemGPT Cloud", + "deployment_type": "Self-hosted (pip, Docker) or Letta Cloud; Letta Code client-side harness (2026)", "tool_support": [ "Memory management", "Tool use", "Custom functions" ], - "pricing_model": "Free open source (MemGPT Cloud available)", - "github_stars": "12000+", + "pricing_model": "Free open source (Letta Cloud managed offering available)", + "github_stars": "23700+", + "github_repo": "https://github.com/letta-ai/letta", + "latest_version": "letta 0.16.8 (May 2026); Letta Code released April 2026", "first_release": "2023", "research_paper": "https://arxiv.org/abs/2310.08560", "storage_backends": "PostgreSQL, SQLite, Chroma", - "rebranding": "Now called Letta (formerly MemGPT)", + "rebranding": "Now called Letta (formerly MemGPT); flagship shifting to Letta Code client-side harness as of March 2026", "pricing": "Free (open source)" }, "use_case_ratings": { diff --git a/data/agents/microsoft-agent-framework.json b/data/agents/microsoft-agent-framework.json index 5c77a56..9ac0630 100644 --- a/data/agents/microsoft-agent-framework.json +++ b/data/agents/microsoft-agent-framework.json @@ -3,8 +3,8 @@ "type": "agent", "name": "Microsoft Agent Framework", "provider": "Microsoft", - "version": "1.0", - "last_evaluated": "2026-06-10", + "version": "1.x", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Open-source SDK and runtime for building AI agents and graph-based multi-agent workflows in .NET and Python. Merges AutoGen and Semantic Kernel into a single framework with checkpointing, middleware, and OpenTelemetry-based observability.", "website": "https://github.com/microsoft/agent-framework", @@ -24,7 +24,7 @@ } ], "methodology": "Assessment of agent execution quality based on framework capabilities, AutoGen/Semantic Kernel lineage, and community-reported results since public preview", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "Review of tool abstraction design, type safety, MCP integration, and middleware-based tool call validation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 87, @@ -52,7 +52,7 @@ } ], "methodology": "Evaluation of graph-based workflow engine for complex task decomposition and deterministic orchestration", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 84, @@ -66,7 +66,7 @@ } ], "methodology": "Review of thread persistence, checkpointing, and memory provider abstractions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -80,7 +80,7 @@ } ], "methodology": "Assessment of checkpoint/resume semantics and middleware-based error handling under failure injection scenarios", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 88, @@ -94,7 +94,7 @@ } ], "methodology": "Multi-agent orchestration pattern review covering group chat, handoff, and concurrent workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of tool execution boundaries in self-hosted and Azure-hosted configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 80, @@ -127,7 +127,7 @@ } ], "methodology": "Review of identity integration, RBAC support, and middleware-based authorization hooks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 74, @@ -141,7 +141,7 @@ } ], "methodology": "Assessment of available guardrail integration points versus out-of-the-box injection protections", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 80, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review of thread isolation and state storage control", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -169,7 +169,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review of framework data handling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 84, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capability assessment for self-hosted and Azure-backed deployments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 78, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis of model provider connectors and telemetry defaults", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 88, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment including local model support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review including migration guides", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 90, @@ -263,7 +263,7 @@ } ], "methodology": "Observability review of built-in OTel spans for agent and workflow execution", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Assessment of workflow visualizability and reasoning trace availability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 94, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment of license and development model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 87, @@ -305,7 +305,7 @@ } ], "methodology": "Community engagement analysis of GitHub activity, discussions, and migration momentum", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment for .NET and Python developers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 85, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability assessment of runtime design and hosting options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 87, @@ -352,7 +352,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 90, @@ -366,7 +366,7 @@ } ], "methodology": "Monitoring features assessment of built-in telemetry", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 85, @@ -377,10 +377,16 @@ "url": "https://github.com/microsoft/agent-framework/releases", "date": "2026-04-03", "value": "1.0 GA released 2026-04-03 with stable API surface; designated successor to Semantic Kernel and AutoGen, which are now in maintenance mode" + }, + { + "source": "Microsoft Agent Framework at BUILD 2026", + "url": "https://devblogs.microsoft.com/agent-framework/microsoft-agent-framework-at-build-2026-announce/", + "date": "2026-07-09", + "value": "Active post-GA cadence: .NET 1.13.0 shipped July 2026 (expanded skills APIs, file editing tools, approval/caching options); Build 2026 announced Agent Harness, Hosted Agents in Foundry Agent Service (GA early July 2026), and CodeAct" } ], "methodology": "Production readiness assessment of GA status, API stability, and Microsoft support commitment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -452,6 +458,8 @@ "Hosted code interpreter (Azure)" ], "first_release": "2025-10-01 (public preview), 1.0 GA 2026-04-03", + "current_version": ".NET 1.13.0 (July 2026); Python on post-GA 1.x cadence", + "github_stars": "11900+", "pricing": "Free (MIT license) - Costs only from model provider and hosting", "predecessors": "Official successor to AutoGen and Semantic Kernel (both maintenance mode)" }, diff --git a/data/agents/n8n-ai-agent.json b/data/agents/n8n-ai-agent.json index b4c4730..f5c5eb6 100644 --- a/data/agents/n8n-ai-agent.json +++ b/data/agents/n8n-ai-agent.json @@ -3,10 +3,10 @@ "type": "agent", "name": "n8n AI Agent", "provider": "n8n", - "version": "1.113.3", - "last_evaluated": "2025-11-09", + "version": "2.29.9", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Fair-code workflow automation platform with AI agent capabilities. Visual workflow builder integrating AI models, tools, and 400+ app integrations for building intelligent automation and agentic workflows.", + "description": "Fair-code workflow automation platform with AI agent capabilities, now on the 2.x major version. Visual builder with AI agent nodes and 400+ integrations. Valued at $5.2B (May 2026) after a $180M Series C (Oct 2025) and a strategic SAP investment. SECURITY: critical CVEs disclosed Jan-Feb 2026 - CVE-2026-21858 'Ni8mare' (unauthenticated RCE), CVE-2026-21877 (CVSS 9.9 RCE), CVE-2026-25049 (authenticated RCE) - all patched; n8n 2.5.2+ is the minimum safe self-hosted version.", "website": "https://n8n.io/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Workflow reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ai_agent_nodes": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "AI agent capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "integration_reliability": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_support": { "score": 84, @@ -66,7 +66,7 @@ } ], "methodology": "LLM integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 86, @@ -80,7 +80,7 @@ } ], "methodology": "Error recovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Variable (100ms-10s)", @@ -94,12 +94,12 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 81, + "overall_score": 77, "criteria": { "credential_management": { "score": 88, @@ -113,7 +113,7 @@ } ], "methodology": "Credential security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_hosting": { "score": 90, @@ -127,10 +127,10 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "authentication": { - "score": 82, + "score": 75, "confidence": "high", "evidence": [ { @@ -138,10 +138,17 @@ "url": "https://docs.n8n.io/user-management/", "date": "2024-10-01", "value": "User authentication with RBAC in enterprise version" + }, + { + "source": "Cyber Centre AL26-001 / Rapid7 - n8n CVE wave", + "url": "https://www.cyber.gc.ca/en/alerts-advisories/al26-001-vulnerabilities-affecting-n8n-cve-2026-21858-cve-2026-21877-cve-2025-68613", + "date": "2026-07-09", + "value": "Jan-Feb 2026 wave of critical vulnerabilities: CVE-2026-21858 'Ni8mare' (unauthenticated RCE, >=1.65.0 <1.121.0), CVE-2026-21877 (CVSS 9.9 RCE via arbitrary file write, 0.123.0-1.121.3), CVE-2026-25049 (authenticated RCE via workflow expressions, fixed in 1.123.17/2.5.2) and eight further CVEs disclosed February 2026; 2.5.2 is the minimum safe version" } ], - "methodology": "Authentication testing", - "last_verified": "2025-11-09" + "methodology": "Authentication testing and security advisory review", + "last_verified": "2026-07-09", + "notes": "Score reduced 2026-07-09: Jan-Feb 2026 wave of critical CVEs including an unauthenticated RCE; all patched (2.5.2+ safe, current release 2.29.9) and no confirmed in-the-wild exploitation reported, hence conservative rather than severe reduction" }, "fair_code_license": { "score": 85, @@ -155,7 +162,7 @@ } ], "methodology": "License assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "webhook_security": { "score": 76, @@ -169,7 +176,7 @@ } ], "methodology": "Webhook security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +195,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 83, @@ -202,7 +209,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 92, @@ -216,7 +223,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_processing": { "score": 80, @@ -230,7 +237,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_logging": { "score": 84, @@ -244,7 +251,7 @@ } ], "methodology": "Logging privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +270,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "visual_workflow_editor": { "score": 94, @@ -277,7 +284,7 @@ } ], "methodology": "UI/UX assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "source_available": { "score": 88, @@ -286,12 +293,12 @@ { "source": "GitHub", "url": "https://github.com/n8n-io/n8n", - "date": "2024-10-20", - "value": "Fair-code license, 48k+ stars, source code available" + "date": "2026-07-09", + "value": "Fair-code license, 195k+ stars, source code available" } ], "methodology": "Source availability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_history": { "score": 86, @@ -305,7 +312,7 @@ } ], "methodology": "Traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 84, @@ -319,7 +326,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +345,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 82, @@ -352,7 +359,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -361,12 +368,12 @@ { "source": "Pricing", "url": "https://n8n.io/pricing/", - "date": "2024-10-01", - "value": "Free self-hosted, cloud pricing starts at $20/month" + "date": "2026-07-09", + "value": "Free self-hosted (fair-code); Cloud Starter from EUR 20/month (annual) or EUR 24/month (monthly) for 2,500 executions; active workflow limits removed across all plans as of April 2026" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 84, @@ -380,7 +387,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "integration_ecosystem": { "score": 95, @@ -394,7 +401,7 @@ } ], "methodology": "Integration ecosystem assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "template_library": { "score": 87, @@ -408,7 +415,7 @@ } ], "methodology": "Template availability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -416,7 +423,7 @@ "strengths": [ "Extremely intuitive visual workflow builder for non-developers", "400+ pre-built integrations with apps and services", - "Fair-code licensed with 48k+ GitHub stars and active community", + "Fair-code licensed with 195k+ GitHub stars and active community", "Self-hosting option for complete data control", "AI agent nodes with LangChain integration for intelligent workflows", "1000+ workflow templates and excellent documentation" @@ -427,7 +434,8 @@ "Can become expensive at scale with cloud hosting", "Learning curve for complex workflow logic", "Limited specialized AI/ML features compared to dedicated platforms", - "Performance can degrade with very complex workflows" + "Performance can degrade with very complex workflows", + "Jan-Feb 2026 wave of critical CVEs (incl. unauthenticated RCE CVE-2026-21858 'Ni8mare' and CVSS 9.9 CVE-2026-21877); self-hosted deployments must run 2.5.2 or later" ], "metadata": { "license": "Sustainable Use License (Fair-code)", @@ -448,11 +456,12 @@ "LangChain tools", "Custom functions" ], - "pricing_model": "Fair-code license - Free self-hosted for personal/internal use, Cloud with unlimited workflows/steps/users (all plans), no active workflow limits as of Aug 2025", - "github_stars": "146956+", - "latest_version": "1.113.3 (September 26, 2025)", - "valuation": "$2.3B (August 2025)", - "annual_revenue": "$40M+", + "pricing_model": "Fair-code license - Free self-hosted for personal/internal use, Cloud Starter from EUR 20/month (annual); unlimited users and no active workflow limits on all plans (confirmed April 2026)", + "github_stars": "195000+", + "latest_version": "2.29.9 (July 9, 2026)", + "valuation": "$5.2B (May 2026, after SAP strategic investment; $2.5B at Oct 2025 $180M Series C led by Accel)", + "total_funding": "$254M", + "annual_revenue": "$40M+ (2025 figure; needs re-verification)", "first_release": "2019", "workflow_templates": "1000+", "ai_features": [ diff --git a/data/agents/openai-agents-sdk.json b/data/agents/openai-agents-sdk.json index 340dcac..23d49ba 100644 --- a/data/agents/openai-agents-sdk.json +++ b/data/agents/openai-agents-sdk.json @@ -3,10 +3,10 @@ "type": "agent", "name": "OpenAI Agents SDK", "provider": "OpenAI", - "version": "1.x", - "last_evaluated": "2026-06-10", + "version": "0.x (Python 0.18.0)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Production-ready multi-agent orchestration framework built around agents, handoffs, guardrails, and tracing. Open-source (MIT) successor to Swarm, released March 2025 with a major overhaul in April 2026.", + "description": "Production-ready multi-agent orchestration framework built around agents, handoffs, guardrails, and tracing. Open-source (MIT) successor to Swarm, released March 2025 with a major overhaul in April 2026 that added a model-native harness (filesystem tools, shell execution, apply-patch edits) and native sandboxed execution for long-horizon tasks.", "website": "https://openai.github.io/openai-agents-python/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Evaluation of agent task success across single- and multi-agent configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 87, @@ -38,7 +38,7 @@ } ], "methodology": "Testing of function tools, hosted tools, and schema-validated argument handling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Multi-step task evaluation across the agent loop and handoff chains", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 78, @@ -66,7 +66,7 @@ } ], "methodology": "Review of session backends and cross-run conversation persistence", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Testing of exception handling, guardrail tripwires, and tool failure paths", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 90, @@ -94,26 +94,33 @@ } ], "methodology": "Multi-agent coordination testing using handoffs and agents-as-tools patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 77, + "overall_score": 79, "criteria": { "tool_sandboxing": { - "score": 65, + "score": 75, "confidence": "medium", "evidence": [ { "source": "OpenAI Agents SDK documentation", "url": "https://openai.github.io/openai-agents-python/tools/", "date": "2026-04-15", - "value": "No built-in sandbox for custom function tools; hosted tools (code interpreter) run in OpenAI's sandboxed environments" + "value": "No built-in sandbox for arbitrary custom function tools; hosted tools (code interpreter) run in OpenAI's sandboxed environments" + }, + { + "source": "OpenAI: The next evolution of the Agents SDK", + "url": "https://openai.com/index/the-next-evolution-of-the-agents-sdk/", + "date": "2026-04-15", + "value": "April 2026 overhaul added native sandbox execution for harness work (files, shell, code edits), launching first in Python with TypeScript to follow" } ], "methodology": "Security architecture review of custom tool execution versus hosted tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09", + "notes": "Score raised 65 to 75: April 2026 release added native sandboxed execution for harness/shell/file operations; custom function tools still execute unsandboxed" }, "access_control": { "score": 70, @@ -127,7 +134,7 @@ } ], "methodology": "Assessment of tool scoping, approval flows, and developer-implemented controls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 80, @@ -141,7 +148,7 @@ } ], "methodology": "Guardrail configuration testing against adversarial and off-policy inputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 75, @@ -155,7 +162,7 @@ } ], "methodology": "Review of run context isolation and self-hosted deployment boundaries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -169,7 +176,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +195,7 @@ } ], "methodology": "Review of OpenAI API retention terms and self-managed session storage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 80, @@ -202,7 +209,7 @@ } ], "methodology": "Compliance capabilities assessment for framework plus default model provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 72, @@ -216,7 +223,7 @@ } ], "methodology": "Data flow analysis of default tracing export and model provider routing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 87, @@ -230,7 +237,7 @@ } ], "methodology": "Deployment options assessment including non-OpenAI and local model routing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +256,7 @@ } ], "methodology": "Documentation completeness review across Python and TypeScript SDKs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 92, @@ -263,7 +270,7 @@ } ], "methodology": "Review of built-in trace spans, dashboard visualization, and OpenTelemetry export", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 84, @@ -277,7 +284,7 @@ } ], "methodology": "Assessment of trace-based explanation of agent routing and tool decisions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 95, @@ -291,7 +298,7 @@ } ], "methodology": "Open source assessment of license, source availability, and public development", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 90, @@ -302,10 +309,16 @@ "url": "https://github.com/openai/openai-agents-python", "date": "2026-06-01", "value": "Tens of thousands of stars, frequent releases, and a large contributor and integration ecosystem" + }, + { + "source": "openai-agents on PyPI", + "url": "https://pypi.org/project/openai-agents/", + "date": "2026-07-09", + "value": "Rapid release cadence continues: Python 0.18.0 released 2026-07-07 (0.17.x through June); repo at ~27,800 stars" } ], "methodology": "Community engagement analysis via GitHub stars, contributor activity, and release cadence", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +337,7 @@ } ], "methodology": "Integration complexity assessment from install to working multi-agent app", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 84, @@ -338,7 +351,7 @@ } ], "methodology": "Assessment of stateless runner scaling and provider rate limit constraints", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 85, @@ -352,7 +365,7 @@ } ], "methodology": "Pricing model analysis of free framework plus pay-per-token model usage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 90, @@ -366,7 +379,7 @@ } ], "methodology": "Monitoring features assessment including built-in and third-party observability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 86, @@ -380,7 +393,7 @@ } ], "methodology": "Maturity assessment from release history, API stability, and enterprise adoption", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -422,10 +435,11 @@ "Free framework; costs limited to model usage" ], "limitations": [ - "No built-in sandboxing for custom function tools; execution safety is developer-owned", + "Custom function tools still run unsandboxed (native sandbox added April 2026 covers harness/shell/file work, Python first)", "Tracing exports run data to OpenAI by default unless explicitly disabled", "Some hosted tools and tracing features work best only with OpenAI models", - "April 2026 overhaul introduced breaking changes requiring migration", + "April 2026 overhaul introduced breaking changes requiring migration (dropped Python 3.9, requires openai v2.x, refusals now raise ModelRefusalError)", + "Python package still versioned 0.x with frequent releases despite production positioning", "Less opinionated about deployment, requiring infrastructure decisions from the team" ], "metadata": { @@ -447,6 +461,8 @@ "Agents as tools" ], "first_release": "2025-03-11; major overhaul 2026-04-15", + "current_version": "Python 0.18.0 (2026-07-07)", + "github_stars": "27700+", "pricing": "Free (MIT); pay only model API rates", "predecessor": "OpenAI Swarm (experimental)" }, diff --git a/data/agents/openai-assistants-api.json b/data/agents/openai-assistants-api.json index 525361d..6088a1d 100644 --- a/data/agents/openai-assistants-api.json +++ b/data/agents/openai-assistants-api.json @@ -4,9 +4,9 @@ "name": "OpenAI Assistants API", "provider": "OpenAI", "version": "v2", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DEPRECATED: the Assistants API will be sunset on 2026-08-26 (roughly 2.5 months away); OpenAI directs users to the Responses API plus Conversations API as the replacement. Previously a managed agent framework with native tool use, code interpreter, file search, and persistent threads. Do not start new projects on it.", + "description": "DEPRECATED: the Assistants API will be sunset on 2026-08-26 (under 7 weeks away); after that date all /v1/assistants, /v1/threads, and /v1/threads/runs calls return errors. OpenAI directs users to the Responses API plus Conversations API as the replacement; Azure OpenAI Assistants retires the same day. Previously a managed agent framework with native tool use, code interpreter, file search, and persistent threads. Do not start new projects on it.", "website": "https://platform.openai.com/docs/assistants/overview", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Based on underlying model performance", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 92, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 87, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 95, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Variable (3-10s typical)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 88, @@ -127,7 +127,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 85, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 90, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 88, @@ -188,7 +188,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "file_handling_security": { "score": 85, @@ -202,7 +202,7 @@ } ], "methodology": "File security testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -221,7 +221,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 80, @@ -235,7 +235,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 75, @@ -249,7 +249,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -271,10 +271,16 @@ "url": "https://developers.openai.com/api/docs/deprecations", "date": "2026-06-10", "value": "Assistants API deprecated with sunset date 2026-08-26; replacement is the Responses API + Conversations API" + }, + { + "source": "OpenAI Developer Community - Assistants API beta deprecation", + "url": "https://community.openai.com/t/assistants-api-beta-deprecation-august-26-2026-sunset/1354666", + "date": "2026-07-09", + "value": "Sunset remains on schedule for 2026-08-26 with no extension or degraded mode; Azure OpenAI Assistants API retires the same date" } ], "methodology": "Integration complexity assessment", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Score reduced: new integrations are discouraged given the 2026-08-26 sunset" }, "scalability": { @@ -289,7 +295,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 82, @@ -303,7 +309,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09", + "last_verified": "2026-07-09", "notes": "Costs can vary significantly based on tool usage" }, "monitoring_capabilities": { @@ -318,7 +324,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -338,7 +344,7 @@ "Vendor lock-in to OpenAI platform", "Limited native explainability features", "Code interpreter limited to Python only", - "Deprecated: sunset scheduled for 2026-08-26; migrate to the Responses API + Conversations API" + "Deprecated: sunset scheduled for 2026-08-26 (verified on track as of 2026-07-09); migrate to the Responses API + Conversations API immediately" ], "metadata": { "license": "Proprietary", diff --git a/data/agents/openai-codex.json b/data/agents/openai-codex.json index f77a387..981b22b 100644 --- a/data/agents/openai-codex.json +++ b/data/agents/openai-codex.json @@ -3,10 +3,10 @@ "type": "agent", "name": "OpenAI Codex", "provider": "OpenAI", - "version": "GPT-5.3-Codex era", - "last_evaluated": "2026-06-10", + "version": "GPT-5.5 era", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's coding agent spanning a cloud agent that runs tasks in isolated containers and an open-source CLI. Delegates parallel software tasks (features, fixes, PRs) powered by GPT-5.3-Codex, with network access disabled by default in the cloud.", + "description": "OpenAI's coding agent spanning a cloud agent that runs tasks in isolated containers and an open-source CLI. Delegates parallel software tasks (features, fixes, PRs) powered by GPT-5.5 (recommended model as of mid-2026; GPT-5.3-Codex deprecated), with network access disabled by default in the cloud.", "website": "https://openai.com/codex/", "trust_vector": { "performance_reliability": { @@ -26,11 +26,11 @@ "source": "GPT-5-Codex upgrade", "url": "https://openai.com/index/introducing-upgrades-to-codex/", "date": "2025-09-15", - "value": "GPT-5-Codex (now GPT-5.3-Codex) substantially improved long-task completion and code quality" + "value": "GPT-5-Codex and successors substantially improved long-task completion and code quality; GPT-5.5 is the recommended Codex model as of mid-2026" } ], "methodology": "Benchmark review and hands-on evaluation of PR-producing cloud tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 87, @@ -44,7 +44,7 @@ } ], "methodology": "Testing of in-container command execution, editing, and test running across repositories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 88, @@ -58,7 +58,7 @@ } ], "methodology": "Long-horizon task evaluation from issue description to passing tests and PR", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 78, @@ -72,7 +72,7 @@ } ], "methodology": "Review of AGENTS.md guidance persistence and per-task container statelessness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -86,7 +86,7 @@ } ], "methodology": "Observed recovery behavior from failing tests, build errors, and missing dependencies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 80, @@ -100,7 +100,7 @@ } ], "methodology": "Assessment of parallel task fan-out and coordination model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -125,7 +125,7 @@ } ], "methodology": "Review of container isolation, default-deny network policy, and CLI sandbox mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 84, @@ -139,7 +139,7 @@ } ], "methodology": "Assessment of repository scoping, environment controls, and approval mode granularity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 85, @@ -153,7 +153,7 @@ } ], "methodology": "Review of network-isolation mitigations and model-level injection defenses", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 88, @@ -167,7 +167,7 @@ } ], "methodology": "Architecture review of per-task container isolation and environment scoping", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 70, @@ -181,7 +181,7 @@ } ], "methodology": "License and source availability review of CLI versus cloud service", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -200,7 +200,7 @@ } ], "methodology": "Review of OpenAI retention and training policies across ChatGPT plan tiers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 80, @@ -214,7 +214,7 @@ } ], "methodology": "Compliance certification and DPA availability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 76, @@ -228,7 +228,7 @@ } ], "methodology": "Data flow analysis of repository access, GitHub integration, and network policy", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 60, @@ -242,7 +242,7 @@ } ], "methodology": "Deployment options assessment of local CLI versus cloud-only agent and models", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -261,7 +261,7 @@ } ], "methodology": "Documentation completeness review across cloud, CLI, and IDE surfaces", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 88, @@ -275,7 +275,7 @@ } ], "methodology": "Review of task logs, test output citations, and diff provenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 84, @@ -289,7 +289,7 @@ } ], "methodology": "Assessment of task summaries, cited reasoning, and pre-merge review surfaces", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 72, @@ -303,7 +303,7 @@ } ], "methodology": "Open source assessment weighting open CLI against proprietary cloud service", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 88, @@ -317,7 +317,7 @@ } ], "methodology": "Community engagement analysis via GitHub activity and release cadence", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -336,7 +336,7 @@ } ], "methodology": "Setup and integration surface assessment across ChatGPT, CLI, IDE, and GitHub", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 90, @@ -350,7 +350,7 @@ } ], "methodology": "Assessment of parallel container execution and plan-tier task throughput", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 75, @@ -364,7 +364,7 @@ } ], "methodology": "Pricing model analysis of plan-based limits, Pro 5x tier, and typical-usage estimates", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 80, @@ -378,7 +378,7 @@ } ], "methodology": "Review of task logs, usage visibility, and admin monitoring features", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 87, @@ -388,11 +388,17 @@ "source": "Codex rollout history", "url": "https://openai.com/index/introducing-codex/", "date": "2025-06-03", - "value": "Research preview 2025-05-16, ChatGPT Plus rollout 2025-06-03, GPT-5-Codex upgrades Sept 2025; now mature on GPT-5.3-Codex" + "value": "Research preview 2025-05-16, ChatGPT Plus rollout 2025-06-03, GPT-5-Codex upgrades Sept 2025" + }, + { + "source": "Codex changelog", + "url": "https://developers.openai.com/codex/changelog", + "date": "2026-07-09", + "value": "Continuous 2026 investment: GPT-5.5 now the recommended Codex model; GPT-5.3-Codex deprecated (no new API requests after 2026-06-30, endpoint shutdown 2026-12-31); June 2026 added Codex Remote GA, Record & Replay skills, and Sites preview" } ], "methodology": "Maturity assessment from rollout timeline, model upgrades, and enterprise availability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -427,7 +433,7 @@ "Verifiable outputs with terminal logs, test results, and action citations", "Open-source Apache-2.0 CLI with local OS-level sandboxing", "Deep GitHub integration from task to reviewed pull request", - "Continuously upgraded models, currently GPT-5.3-Codex" + "Continuously upgraded models, currently GPT-5.5 (with GPT-5.4 / GPT-5.4 mini options)" ], "limitations": [ "Default network isolation can block tasks needing external dependencies unless allowlists are configured", @@ -439,7 +445,9 @@ "metadata": { "license": "Cloud agent proprietary; Codex CLI Apache-2.0 (github.com/openai/codex)", "supported_models": [ - "GPT-5.3-Codex (current)", + "GPT-5.5 (recommended, mid-2026)", + "GPT-5.4 / GPT-5.4 mini", + "GPT-5.3-Codex (deprecated; API sunset 2026-12-31)", "GPT-5-Codex (Sept 2025)", "codex-1 (launch)" ], diff --git a/data/agents/pydantic-ai.json b/data/agents/pydantic-ai.json index 1bedf6c..0981a66 100644 --- a/data/agents/pydantic-ai.json +++ b/data/agents/pydantic-ai.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Pydantic AI", "provider": "Pydantic", - "version": "1.12.0", - "last_evaluated": "2026-06-10", + "version": "2.7.0", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Type-safe Python agent framework from the creators of Pydantic. Reached v1.0 stable on 2025-09-04 with a formal API stability commitment and has surpassed 15M+ downloads. Provides production-ready agents with strong typing, validation, and structured outputs, designed for reliability and maintainability in production systems.", + "description": "Type-safe Python agent framework from the creators of Pydantic. Reached v1.0 stable on 2025-09-04 with a formal API stability commitment, then v2.0 on 2026-06-23 introducing a harness-first design with 'capabilities' (composable bundles of tools, hooks, instructions, and model settings) as a core primitive. Provides production-ready agents with strong typing, validation, and structured outputs, designed for reliability and maintainability in production systems.", "website": "https://ai.pydantic.dev/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Type safety testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "structured_outputs": { "score": 92, @@ -38,7 +38,7 @@ } ], "methodology": "Output validation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_integration": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "LLM integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "validation_reliability": { "score": 90, @@ -66,7 +66,7 @@ } ], "methodology": "Validation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_calling": { "score": 84, @@ -80,7 +80,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Variable (LLM-dependent)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "type_safety_security": { "score": 90, @@ -127,7 +127,7 @@ } ], "methodology": "Security testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_hosting": { "score": 88, @@ -141,7 +141,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 92, @@ -155,7 +155,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "dependency_security": { "score": 68, @@ -169,7 +169,7 @@ } ], "methodology": "Dependency analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 75, @@ -202,7 +202,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 90, @@ -216,7 +216,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 82, @@ -230,7 +230,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "no_telemetry": { "score": 88, @@ -244,7 +244,7 @@ } ], "methodology": "Telemetry assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "type_hints": { "score": 95, @@ -277,7 +277,7 @@ } ], "methodology": "Developer experience assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 92, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "validation_errors": { "score": 88, @@ -305,7 +305,7 @@ } ], "methodology": "Error messaging assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_trust": { "score": 84, @@ -319,7 +319,7 @@ } ], "methodology": "Community trust assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 80, @@ -352,7 +352,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 90, @@ -366,7 +366,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 78, @@ -380,7 +380,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 90, @@ -397,11 +397,17 @@ "url": "https://pydantic.dev/articles/pydantic-ai-v1", "date": "2026-06-10", "value": "v1.0 stable released 2025-09-04 with API stability commitment; 15M+ downloads" + }, + { + "source": "Pydantic AI Changelog / Upgrade Guide", + "url": "https://pydantic.dev/docs/ai/project/changelog/", + "date": "2026-07-09", + "value": "v2.0.0 released 2026-06-23 (harness-first design, capabilities primitive, breaking changes from v1); latest release v2.7.0 on 2026-07-09" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10", - "notes": "Score raised: v1.0 stable release with API stability commitment confirms production maturity" + "last_verified": "2026-07-09", + "notes": "Score raised: v1.0 stable release with API stability commitment confirms production maturity. v2.0 (2026-06-23) is a major release with breaking changes; teams on v1 need a migration, but release cadence and repo activity remain very high (last push 2026-07-09)" }, "testing_support": { "score": 86, @@ -415,7 +421,7 @@ } ], "methodology": "Testing capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -435,7 +441,7 @@ "Requires Python and Pydantic knowledge", "No built-in monitoring or observability tools", "Less opinionated than full-featured frameworks", - "Status (2026-06): actively developed; v1.0 stable since 2025-09-04 with API stability commitment, so pre-1.0 API churn concerns no longer apply" + "Status (2026-07): actively developed; v2.0 major release (2026-06-23) introduced breaking changes (capabilities restructuring, renamed model classes, removed A2A/FastMCP/Outlines integrations), so v1 users face a migration; latest v2.7.0 (2026-07-09)" ], "metadata": { "license": "MIT", @@ -464,9 +470,10 @@ ], "pricing_model": "Free open source (MIT license)", "first_release": "2024", - "github_stars": "13236+", + "github_stars": "18200+", "v1_release": "September 2025 (API stability commitment)", - "latest_version": "v1.12.0 (November 7, 2025)", + "v2_release": "June 2026 (v2.0.0 on 2026-06-23; harness-first design with capabilities primitive)", + "latest_version": "v2.7.0 (July 9, 2026)", "parent_project": "Pydantic (70M+ downloads/month)", "github_repo": "https://github.com/pydantic/pydantic-ai", "key_features": [ diff --git a/data/agents/rasa.json b/data/agents/rasa.json index ef9fa1a..3468753 100644 --- a/data/agents/rasa.json +++ b/data/agents/rasa.json @@ -1,13 +1,13 @@ { "id": "rasa", "type": "agent", - "name": "Rasa Open Source", + "name": "Rasa (Open Source & Rasa Pro)", "provider": "Rasa Technologies", - "version": "3.x", - "last_evaluated": "2025-11-09", + "version": "Rasa Pro 3.16.x; Rasa Open Source 3.x (maintenance mode)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Leading open-source conversational AI framework for building contextual assistants and chatbots. Provides full control over NLU, dialogue management, and deployment with on-premises and cloud options.", - "website": "https://rasa.com/docs/rasa/", + "description": "Open-source-rooted conversational AI framework for building contextual assistants with full control over NLU, dialogue management, and deployment. NOTE: classic Rasa Open Source (Apache 2.0) is in maintenance mode; Rasa's strategic direction is CALM, its LLM-native dialogue approach, delivered via the commercially licensed Rasa Pro (3.16.x as of mid-2026) and the no-code Rasa Studio. A free Rasa Pro Developer Edition covers 1,000 conversations/month; paid tiers start at ~$35,000/year.", + "website": "https://rasa.com/docs/pro/intro/", "trust_vector": { "performance_reliability": { "overall_score": 82, @@ -24,7 +24,7 @@ } ], "methodology": "Intent accuracy benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "dialogue_management": { "score": 87, @@ -38,7 +38,7 @@ } ], "methodology": "Conversation flow testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "entity_extraction": { "score": 83, @@ -52,7 +52,7 @@ } ], "methodology": "Entity extraction testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "custom_actions": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Custom action testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "context_handling": { "score": 79, @@ -80,7 +80,7 @@ } ], "methodology": "Context retention testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "100-500ms (self-hosted)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "authentication": { "score": 75, @@ -127,7 +127,7 @@ } ], "methodology": "Authentication capabilities review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_privacy": { "score": 85, @@ -141,7 +141,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -155,7 +155,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "encryption": { "score": 62, @@ -169,7 +169,7 @@ } ], "methodology": "Security features review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 88, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "on_premise_deployment": { "score": 95, @@ -216,7 +216,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -230,7 +230,7 @@ } ], "methodology": "PII protection assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "no_data_sharing": { "score": 95, @@ -244,7 +244,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 95, @@ -274,10 +274,16 @@ "url": "https://github.com/RasaHQ/rasa", "date": "2024-10-20", "value": "Apache 2.0, 18k+ stars, active community contributions" + }, + { + "source": "Rasa documentation (product split)", + "url": "https://rasa.com/docs/pro/intro/", + "date": "2026-07-09", + "value": "Rasa Open Source is in maintenance mode; active development has moved to CALM in the commercially licensed Rasa Pro (3.16.x) and Rasa Studio; legacy OSS remains Apache 2.0" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversation_debugging": { "score": 82, @@ -291,7 +297,7 @@ } ], "methodology": "Debugging tools assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 85, @@ -305,7 +311,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "explainability": { "score": 78, @@ -319,7 +325,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +344,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 76, @@ -352,7 +358,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 95, @@ -363,10 +369,16 @@ "url": "https://github.com/RasaHQ/rasa", "date": "2024-10-01", "value": "Free open source, costs only for infrastructure" + }, + { + "source": "Rasa Pro licensing", + "url": "https://rasa.com/docs/rasa-pro/installation/python/licensing/", + "date": "2026-07-09", + "value": "Free Rasa Pro Developer Edition (up to 1,000 conversations/month, 100/month for internal employee agents); Growth tier from ~$35,000/year; Enterprise custom; legacy OSS remains free (Apache 2.0) but in maintenance mode" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 73, @@ -380,7 +392,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 74, @@ -394,7 +406,7 @@ } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "training_pipeline": { "score": 81, @@ -408,13 +420,13 @@ } ], "methodology": "Training capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } }, "strengths": [ - "Open source (Apache 2.0) with complete control over code and data", + "Open-source roots (Apache 2.0 legacy) with complete control over code and data; CALM in Rasa Pro adds LLM-native dialogue while keeping business logic deterministic", "On-premises deployment option for maximum privacy and compliance", "No vendor lock-in or usage-based pricing, only infrastructure costs", "Highly customizable NLU pipeline and dialogue policies", @@ -422,6 +434,7 @@ "Machine learning-based dialogue management for contextual conversations" ], "limitations": [ + "Rasa Open Source is in maintenance mode: new capabilities (CALM, LLM-native dialogue, Rasa Studio) ship only in the commercially licensed Rasa Pro", "Requires significant ML and DevOps expertise for deployment", "Limited enterprise features out-of-box (auth, monitoring, analytics)", "Steeper learning curve compared to managed cloud services", @@ -430,11 +443,12 @@ "Training data quality critical for good performance" ], "metadata": { - "license": "Apache 2.0", + "license": "Apache 2.0 (legacy Rasa Open Source, maintenance mode); Rasa Pro is commercially licensed", "supported_models": [ "DIET", "TED Policy", - "Transformer Embedding Dialogue" + "Transformer Embedding Dialogue", + "CALM with pluggable LLMs (Rasa Pro)" ], "programming_languages": [ "Python" @@ -445,13 +459,13 @@ "REST API", "Webhooks" ], - "pricing_model": "Free open source (Rasa Pro available for enterprise)", + "pricing_model": "Legacy OSS free (maintenance mode); Rasa Pro: free Developer Edition, Growth from ~$35,000/year, Enterprise custom", "github_stars": "20700+", "first_release": "2016", "supported_channels": "Slack, Facebook, Telegram, Web, Custom REST", - "enterprise_version": "Rasa Pro (commercial license available)", - "pricing": "Free Developer Edition (1,000 conversations/month), Entry-level from $100-500, Rasa Pro from $35,000+", - "version": "3.x" + "enterprise_version": "Rasa Pro with CALM + Rasa Studio (commercial license)", + "pricing": "Free Rasa Pro Developer Edition (1,000 conversations/month; 100/month for internal employee agents), Growth tier from ~$35,000/year, Enterprise custom", + "version": "Rasa Pro 3.16.x; Rasa Open Source 3.x (maintenance mode)" }, "use_case_ratings": { "customer-support": { diff --git a/data/agents/relevance-ai.json b/data/agents/relevance-ai.json index 9dade83..3d455ff 100644 --- a/data/agents/relevance-ai.json +++ b/data/agents/relevance-ai.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Relevance AI", "provider": "Relevance AI Pty Ltd", - "version": "2025.1", - "last_evaluated": "2025-01-14", + "version": "Current (SaaS)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "No-code AI agent platform that enables teams to build, customize, and deploy AI workforce agents. Features visual workflow builders, tool integrations, and the ability to create specialized AI employees for various business functions including sales, support, and research.", + "description": "No-code AI agent platform that enables teams to build, customize, and deploy AI workforce agents. Features visual workflow builders, tool integrations, and the ability to create specialized AI employees for various business functions including sales, support, and research. Raised a $24M Series B (May 2025, led by Bessemer) on the back of rapid growth (40,000 agents registered in January 2025 alone); pricing now splits usage into Actions and Vendor Credits, with BYO LLM API keys on paid plans.", "website": "https://relevanceai.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Task completion testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "tool_integration": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "workflow_reliability": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Workflow reliability testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "model_flexibility": { "score": 85, @@ -66,7 +66,7 @@ } ], "methodology": "Model flexibility review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "error_handling": { "score": 78, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Security certification review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "api_key_management": { "score": 78, @@ -113,7 +113,7 @@ } ], "methodology": "Key management review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "action_permissions": { "score": 75, @@ -127,7 +127,7 @@ } ], "methodology": "Permission model review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 78, @@ -141,7 +141,7 @@ } ], "methodology": "Audit capability review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -160,7 +160,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "third_party_llm_exposure": { "score": 72, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "score": 78, @@ -188,7 +188,7 @@ } ], "methodology": "Retention policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -207,7 +207,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "agent_visibility": { "score": 85, @@ -221,7 +221,7 @@ } ], "methodology": "Visibility assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "pricing_transparency": { "score": 82, @@ -230,12 +230,12 @@ { "source": "Relevance AI Pricing", "url": "https://relevanceai.com/pricing", - "date": "2025-01-10", - "value": "Clear pricing tiers with credit-based system" + "date": "2026-07-09", + "value": "Clear pricing tiers; usage split into Actions (task runs) and Vendor Credits (AI compute), Vendor Credits roll over indefinitely" } ], "methodology": "Pricing review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "community_engagement": { "score": 80, @@ -249,7 +249,7 @@ } ], "methodology": "Community engagement review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -268,7 +268,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "template_library": { "score": 85, @@ -282,7 +282,7 @@ } ], "methodology": "Template library review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "deployment_speed": { "score": 88, @@ -296,7 +296,7 @@ } ], "methodology": "Deployment speed testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "cost_efficiency": { "score": 80, @@ -305,12 +305,12 @@ { "source": "Relevance AI Pricing", "url": "https://relevanceai.com/pricing", - "date": "2025-01-10", - "value": "Credit-based pricing with free tier available" + "date": "2026-07-09", + "value": "Free tier (200 Actions/month), Pro from $19/month (annual, 2,500 Actions), Team $234/month (annual; $349 monthly, 7,000 Actions); Pro+ plans support bring-your-own OpenAI/Anthropic/Google API keys, bypassing Vendor Credits" } ], "methodology": "Cost analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "customization": { "score": 82, @@ -324,7 +324,7 @@ } ], "methodology": "Customization review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } @@ -397,10 +397,11 @@ "license": "Proprietary", "supported_platforms": ["Cloud"], "deployment_type": "SaaS", - "pricing": "Free tier, Pro from $19/month, Team from $199/month", + "pricing": "Free: 200 Actions/month; Pro from $19/month (annual); Team $234/month (annual, $349 monthly); Enterprise custom (verified 2026-07-09)", "certifications": ["SOC 2 Type II"], "founded": "2020", "headquarters": "Sydney, Australia", + "funding": "Series B $24M (May 2025, led by Bessemer Venture Partners; total raised $37M)", "supported_llms": ["OpenAI", "Anthropic", "Google", "Cohere"] }, "tags": ["no-code", "workflow-automation", "sales", "marketing", "research"] diff --git a/data/agents/salesforce-einstein-bots.json b/data/agents/salesforce-einstein-bots.json index 7a94ba6..5d98cd5 100644 --- a/data/agents/salesforce-einstein-bots.json +++ b/data/agents/salesforce-einstein-bots.json @@ -4,9 +4,9 @@ "name": "Salesforce Einstein Bots", "provider": "Salesforce", "version": "Einstein GPT", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "TRANSITIONING: Einstein Copilot was retired and absorbed into Agentforce (Jan 2025, Spring '25 release), and Salesforce is steering Einstein Bots customers toward Agentforce agents. Einstein Bots remains Salesforce's CRM-integrated chatbot platform, combining Einstein AI with CRM data for personalized interactions and human-agent handoffs.", + "description": "TRANSITIONING: Einstein Copilot was retired into Agentforce (Jan 2025, Spring '25 release) and Salesforce is steering Einstein Bots customers toward Agentforce. Einstein Bots has no announced end-of-life as of mid-2026 but is effectively legacy: Legacy Chat was fully retired February 14, 2026 (bots must run on Messaging for In-App and Web), Article Answers was retired December 31, 2025, and a 'Create AI Agents from Einstein Bots' scaffolding tool aids migration to Agentforce agents.", "website": "https://www.salesforce.com/products/service-cloud/features/einstein-bots/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Intent accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "crm_integration": { "score": 95, @@ -38,7 +38,7 @@ } ], "methodology": "CRM integration assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "dialog_flow": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Conversation design testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "handoff_to_agent": { "score": 92, @@ -66,7 +66,7 @@ } ], "methodology": "Handoff workflow testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "personalization": { "score": 88, @@ -80,7 +80,7 @@ } ], "methodology": "Personalization capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "400-900ms average", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 92, @@ -127,7 +127,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_governance": { "score": 90, @@ -141,7 +141,7 @@ } ], "methodology": "Data governance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_trail": { "score": 89, @@ -155,7 +155,7 @@ } ], "methodology": "Audit capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance": { "score": 91, @@ -169,7 +169,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 91, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "hipaa_compliance": { "score": 90, @@ -216,7 +216,7 @@ } ], "methodology": "Healthcare compliance review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 87, @@ -230,7 +230,7 @@ } ], "methodology": "PII protection testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_sovereignty": { "score": 89, @@ -244,7 +244,7 @@ } ], "methodology": "Data residency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "analytics": { "score": 86, @@ -277,7 +277,7 @@ } ], "methodology": "Analytics capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "testing_tools": { "score": 81, @@ -291,7 +291,7 @@ } ], "methodology": "Testing features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "explainability": { "score": 78, @@ -305,7 +305,7 @@ } ], "methodology": "Explainability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -330,7 +330,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 91, @@ -344,7 +344,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 80, @@ -358,7 +358,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 87, @@ -372,7 +372,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "channel_support": { "score": 90, @@ -383,10 +383,16 @@ "url": "https://help.salesforce.com/s/articleView?id=sf.messaging_overview.htm", "date": "2024-10-01", "value": "SMS, WhatsApp, Facebook Messenger, Web Chat, Slack" + }, + { + "source": "Salesforce Legacy Chat retirement", + "url": "https://help.salesforce.com/s/articleView?id=005134861&language=en_US&type=1", + "date": "2026-07-09", + "value": "Legacy Chat fully retired February 14, 2026 - Einstein Bots must run on Messaging for In-App and Web; Article Answers retired December 31, 2025" } ], "methodology": "Channel integration assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "workflow_automation": { "score": 89, @@ -400,7 +406,7 @@ } ], "methodology": "Automation capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -420,7 +426,7 @@ "Not suitable for organizations not using Salesforce CRM", "Conversation design less sophisticated than specialized platforms", "Einstein AI capabilities lag behind latest LLM-based systems", - "Product direction has shifted to Agentforce: Einstein Copilot retired (Jan 2025) and Einstein Bots are being steered toward Agentforce agents" + "Product direction shifted to Agentforce (Einstein Copilot retired Jan 2025); Legacy Chat fully retired February 14, 2026 and Article Answers December 31, 2025, and migration to Agentforce is a rebuild - the 'Create AI Agents from Einstein Bots' tool only scaffolds Topics/Actions from existing intents and dialogs" ], "metadata": { "license": "Proprietary", diff --git a/data/agents/semantic-kernel-agent.json b/data/agents/semantic-kernel-agent.json index bf88037..abd59c5 100644 --- a/data/agents/semantic-kernel-agent.json +++ b/data/agents/semantic-kernel-agent.json @@ -4,7 +4,7 @@ "name": "Semantic Kernel Agent", "provider": "Microsoft", "version": "1.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "MAINTENANCE MODE: Semantic Kernel now receives bug/security fixes only and is superseded by the Microsoft Agent Framework (1.0 GA on 2026-04-03), the recommended migration path. It remains Microsoft's enterprise SDK for integrating LLMs with conventional languages via plugins, planners, and memory, with .NET, Python, and Java support.", "website": "https://learn.microsoft.com/en-us/semantic-kernel/", @@ -24,7 +24,7 @@ } ], "methodology": "Based on enterprise testing and model performance", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 91, @@ -38,7 +38,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 86, @@ -52,7 +52,7 @@ } ], "methodology": "Complex task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 89, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Low (1-5s typical)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "access_control": { "score": 90, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 87, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 92, @@ -155,7 +155,7 @@ } ], "methodology": "Data architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "enterprise_security": { "score": 93, @@ -169,7 +169,7 @@ } ], "methodology": "Enterprise security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 93, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 88, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 88, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 85, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 93, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "enterprise_support": { "score": 92, @@ -305,7 +305,7 @@ } ], "methodology": "Support options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 93, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -352,7 +352,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 94, @@ -366,7 +366,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 92, @@ -383,10 +383,16 @@ "url": "https://devblogs.microsoft.com/semantic-kernel/migrate-your-semantic-kernel-and-autogen-projects-to-microsoft-agent-framework-release-candidate/", "date": "2026-06-10", "value": "Semantic Kernel is in maintenance mode (bug/security fixes only); Microsoft Agent Framework 1.0 reached GA on 2026-04-03 as the successor" + }, + { + "source": "Microsoft Agent Framework Overview (Microsoft Learn)", + "url": "https://learn.microsoft.com/en-us/agent-framework/overview/", + "date": "2026-07-09", + "value": "Maintenance-mode status still stands: Microsoft directs new work to Agent Framework with official Semantic Kernel migration guide; SK repo (~28,300 stars) still receives bug/security fixes" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -428,7 +434,7 @@ "Native functions", "Semantic functions" ], - "github_stars": "25800+", + "github_stars": "28200+", "first_release": "2023", "ga_version": "1.45 (.NET) and 1.27 (Python) - April 2025", "transition": "Microsoft Agent Framework is the successor (Semantic Kernel v2.0)" diff --git a/data/agents/sierra-ai.json b/data/agents/sierra-ai.json index b5d1fe0..8e23044 100644 --- a/data/agents/sierra-ai.json +++ b/data/agents/sierra-ai.json @@ -3,10 +3,10 @@ "type": "agent", "name": "Sierra", "provider": "Sierra Technologies Inc.", - "version": "2025.1", - "last_evaluated": "2025-01-14", + "version": "Agent OS 2.0 (2026)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Conversational AI platform specialized in autonomous customer experience agents. Founded by former Salesforce co-CEO Bret Taylor and ex-Google executive Clay Bavor, Sierra focuses on transactional CX workflows with agents that can take real actions on behalf of customers.", + "description": "Conversational AI platform for autonomous customer experience agents, built on Sierra's Agent OS 2.0 spanning chat, voice, email, SMS, and WhatsApp. Founded by Bret Taylor (ex-Salesforce co-CEO) and Clay Bavor (ex-Google), Sierra focuses on CX agents that take real actions. One of the most heavily funded AI agent startups: $350M at a $10B valuation (Sep 2025, Greenoaks), then a $950M Series E at $15.8B (May 2026, Tiger Global and GV); 40%+ of the Fortune 50 are customers, ARR $150M+.", "website": "https://sierra.ai/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Conversation quality testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "action_execution": { "score": 88, @@ -35,10 +35,16 @@ "url": "https://sierra.ai/", "date": "2025-01-10", "value": "Agents can take real actions: process returns, update orders, etc." + }, + { + "source": "Sierra Agent OS 2.0", + "url": "https://sierra.ai/blog/agent-os-2-0", + "date": "2026-07-09", + "value": "Agent OS 2.0 adds persistent memory and expanded action execution across chat, voice, email, SMS, and WhatsApp channels" } ], "methodology": "Action capability testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "resolution_rate": { "score": 85, @@ -52,7 +58,7 @@ } ], "methodology": "Resolution rate analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "brand_alignment": { "score": 88, @@ -66,7 +72,7 @@ } ], "methodology": "Brand alignment testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "escalation_handling": { "score": 85, @@ -80,7 +86,7 @@ } ], "methodology": "Escalation flow testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -99,7 +105,7 @@ } ], "methodology": "Security assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "action_guardrails": { "score": 85, @@ -113,7 +119,7 @@ } ], "methodology": "Guardrails review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_protection": { "score": 85, @@ -127,7 +133,7 @@ } ], "methodology": "Data protection review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "access_control": { "score": 85, @@ -141,7 +147,7 @@ } ], "methodology": "Access control review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -160,7 +166,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "compliance_support": { "score": 82, @@ -174,7 +180,7 @@ } ], "methodology": "Compliance review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_minimization": { "score": 80, @@ -188,7 +194,7 @@ } ], "methodology": "Data practices review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -207,7 +213,7 @@ } ], "methodology": "Visibility assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "action_transparency": { "score": 82, @@ -221,7 +227,7 @@ } ], "methodology": "Action logging review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "model_disclosure": { "score": 72, @@ -235,7 +241,7 @@ } ], "methodology": "Technology transparency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "company_transparency": { "score": 82, @@ -246,10 +252,16 @@ "url": "https://sierra.ai/", "date": "2025-01-10", "value": "Well-known founders with public profiles" + }, + { + "source": "TechCrunch - Sierra Series E", + "url": "https://techcrunch.com/2026/05/04/sierra-raises-950m-as-the-race-to-own-enterprise-ai-gets-serious/", + "date": "2026-07-09", + "value": "$950M Series E (May 2026) at a $15.8B post-money valuation led by Tiger Global and GV, following $350M at $10B (Sep 2025, Greenoaks); 40%+ of the Fortune 50 are customers with $150M+ reported ARR" } ], "methodology": "Company transparency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -268,7 +280,7 @@ } ], "methodology": "Implementation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "system_integration": { "score": 85, @@ -282,7 +294,7 @@ } ], "methodology": "Integration review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "analytics": { "score": 85, @@ -296,7 +308,7 @@ } ], "methodology": "Analytics capability review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "continuous_improvement": { "score": 82, @@ -310,7 +322,7 @@ } ], "methodology": "Improvement process review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } @@ -374,7 +386,7 @@ "limitations": [ "Focused solely on customer service use cases", "Enterprise pricing - not for small businesses", - "Newer company (founded 2023)", + "Young company (founded 2023), though now heavily capitalized ($15.8B valuation, May 2026) with large enterprise traction", "Limited public documentation", "Proprietary technology with limited disclosure", "Requires system integrations for full capability" @@ -383,12 +395,14 @@ "license": "Proprietary", "supported_platforms": ["Cloud"], "deployment_type": "SaaS", - "pricing": "Enterprise pricing - contact sales", + "pricing": "Enterprise pricing - contact sales (known for outcome-based, pay-per-resolution pricing)", "founded": "2023", "founders": "Bret Taylor (ex-Salesforce co-CEO), Clay Bavor (ex-Google)", "headquarters": "San Francisco, CA", - "funding": "$285M Series C at $4.5B valuation (Dec 2024)", - "customers": "WeightWatchers, SiriusXM, ADT, OluKai" + "funding": "$950M Series E at $15.8B valuation (May 2026, led by Tiger Global and GV); previously $350M at $10B (Sep 2025, Greenoaks) and $285M Series C at $4.5B (Dec 2024)", + "customers": "40%+ of the Fortune 50; WeightWatchers, SiriusXM, ADT, OluKai, Minted, plus major banks, fintechs, and healthcare organizations", + "channels": ["Chat", "Voice", "Email", "SMS", "WhatsApp"], + "arr": "$150M+ reported (2026)" }, "tags": ["customer-support", "conversational-ai", "autonomous-agents", "e-commerce"] } diff --git a/data/agents/smolagents.json b/data/agents/smolagents.json index 468c0c5..638f0de 100644 --- a/data/agents/smolagents.json +++ b/data/agents/smolagents.json @@ -4,7 +4,7 @@ "name": "smolagents", "provider": "Hugging Face", "version": "1.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Minimalist Python agent library from Hugging Face. Its signature CodeAgent writes actions as executable Python code instead of JSON tool calls, enabling expressive multi-step behavior with a deliberately small core codebase.", "website": "https://github.com/huggingface/smolagents", @@ -24,7 +24,7 @@ } ], "methodology": "Review of published benchmark comparisons between code actions and JSON tool calls, plus task completion testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "Tool invocation testing across CodeAgent and ToolCallingAgent modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Complex multi-step task testing using ReAct loop with planning interval enabled", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 65, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation across single-run and cross-session scenarios", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 75, @@ -80,7 +80,7 @@ } ], "methodology": "Error injection testing observing self-correction behavior across retries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 74, @@ -94,7 +94,7 @@ } ], "methodology": "Multi-agent coordination testing using managed agents hierarchy", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of the local Python executor versus opt-in remote sandbox backends", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 60, @@ -127,7 +127,7 @@ } ], "methodology": "Access control capabilities assessment of library surface", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 58, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack surface review; code-action paradigm amplifies impact of successful injection", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 62, @@ -155,7 +155,7 @@ } ], "methodology": "Data isolation architecture review across executor backends", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -169,7 +169,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review of self-hosted library model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 80, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment across deployment configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 75, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis across model and executor backends", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 92, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment including air-gapped configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 82, @@ -263,7 +263,7 @@ } ], "methodology": "Tracing and logging capabilities assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability assessment of agent step outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 95, @@ -291,7 +291,7 @@ } ], "methodology": "Open source assessment of license, codebase size, and auditability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 88, @@ -305,7 +305,7 @@ } ], "methodology": "Community engagement analysis of commits, issues, and releases", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Integration complexity assessment with minimal-setup testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 72, @@ -338,7 +338,7 @@ } ], "methodology": "Scalability architecture assessment for production workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -352,7 +352,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 76, @@ -366,7 +366,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 70, @@ -377,10 +377,16 @@ "url": "https://github.com/huggingface/smolagents/releases", "date": "2026-05-30", "value": "Actively developed with occasional breaking changes; production deployments must add sandboxing, scaling, and guardrails themselves" + }, + { + "source": "smolagents GitHub Releases", + "url": "https://github.com/huggingface/smolagents/releases", + "date": "2026-07-09", + "value": "Latest release v1.26.0 (2026-05-29); repo active (last push 2026-06-23) at ~28,300 stars; still on 1.x series, no 2.0 release" } ], "methodology": "Production readiness assessment of API stability and operational gaps", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -461,8 +467,9 @@ "MCP tools", "LangChain tool import" ], - "github_stars": "20000+", + "github_stars": "28200+", "first_release": "2024 (Dec 2024/Jan 2025)", + "current_version": "1.26.0 (2026-05-29)", "pricing": "Free (Apache 2.0) - Costs only from LLM API calls and optional sandbox providers", "python_requirement": "Python >=3.10", "adoption": "One of the most-starred minimalist agent libraries; widely used for open deep-research agents" diff --git a/data/agents/strands-agents.json b/data/agents/strands-agents.json index 510ec65..3897917 100644 --- a/data/agents/strands-agents.json +++ b/data/agents/strands-agents.json @@ -4,9 +4,9 @@ "name": "Strands Agents", "provider": "Amazon Web Services", "version": "1.x", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source, model-driven AI agent SDK from AWS, used internally by Amazon Q Developer. Takes a lightweight model-first approach with MCP and A2A support, multi-agent primitives, and optional pairing with Amazon Bedrock AgentCore for hosted runtime.", + "description": "Open-source, model-driven AI agent SDK from AWS, used internally by Amazon Q Developer. Takes a lightweight model-first approach with MCP and A2A support, multi-agent primitives, and optional pairing with Amazon Bedrock AgentCore for hosted runtime. In June 2026 the core repo was consolidated into the strands-agents/harness-sdk monorepo (Python + TypeScript), repositioned around AWS's 'agent harness' framing; package names (strands-agents on PyPI) are unchanged.", "website": "https://strandsagents.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Task completion testing plus review of documented internal AWS production usage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_use_reliability": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Tool invocation testing across native tools and MCP servers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "multi_step_planning": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Complex multi-step task testing with agent loop and graph patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "memory_persistence": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Memory system evaluation across sessions and AgentCore Memory integration", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 79, @@ -80,7 +80,7 @@ } ], "methodology": "Error injection testing observing model-driven recovery", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "agent_collaboration": { "score": 86, @@ -94,7 +94,7 @@ } ], "methodology": "Multi-agent coordination testing across swarm, graph, and A2A patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review of SDK plus AgentCore runtime isolation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 78, @@ -127,7 +127,7 @@ } ], "methodology": "Access control assessment of IAM and identity integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_defense": { "score": 72, @@ -141,7 +141,7 @@ } ], "methodology": "Injection testing with and without Bedrock Guardrails enabled", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_isolation": { "score": 78, @@ -155,7 +155,7 @@ } ], "methodology": "Data isolation architecture review across deployment targets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -169,7 +169,7 @@ } ], "methodology": "Source code and governance review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review of SDK and managed runtime options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 84, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment leveraging AWS compliance posture", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 78, @@ -216,7 +216,7 @@ } ], "methodology": "Data flow analysis across supported model providers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_deployment_option": { "score": 88, @@ -230,7 +230,7 @@ } ], "methodology": "Deployment options assessment including local model configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "execution_traceability": { "score": 88, @@ -263,7 +263,7 @@ } ], "methodology": "Tracing and telemetry capabilities assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Explainability assessment of loop transparency and hook system", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 92, @@ -288,10 +288,16 @@ "url": "https://github.com/strands-agents", "date": "2026-06-01", "value": "Apache 2.0 SDK, tools, and samples; Python SDK 1.0 released 2026-05-21, TypeScript 1.0 released 2026-04-30" + }, + { + "source": "strands-agents/harness-sdk GitHub", + "url": "https://github.com/strands-agents/harness-sdk", + "date": "2026-07-09", + "value": "Core repo renamed from sdk-python to harness-sdk (June 2026) as an Apache 2.0 monorepo shipping Python and TypeScript SDKs, CLI, and docs; old sdk-python URLs redirect" } ], "methodology": "Open source assessment of license and release maturity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_activity": { "score": 85, @@ -305,7 +311,7 @@ } ], "methodology": "Community engagement analysis of downloads, releases, and contributors", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -324,7 +330,7 @@ } ], "methodology": "Integration complexity assessment with minimal-setup testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scalability": { "score": 86, @@ -338,7 +344,7 @@ } ], "methodology": "Scalability assessment across managed and self-managed deployment targets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 84, @@ -352,7 +358,7 @@ } ], "methodology": "Pricing model analysis of SDK and optional managed services", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_capabilities": { "score": 86, @@ -366,7 +372,7 @@ } ], "methodology": "Monitoring features assessment across SDK and AWS integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 87, @@ -374,13 +380,19 @@ "evidence": [ { "source": "Strands 1.0 Releases / AWS Internal Usage", - "url": "https://github.com/strands-agents/sdk-python/releases", + "url": "https://github.com/strands-agents/harness-sdk/releases", "date": "2026-05-21", "value": "Python 1.0 (2026-05-21) and TypeScript 1.0 (2026-04-30) with semver stability; battle-tested in Amazon Q Developer" + }, + { + "source": "PyPI strands-agents / harness-sdk releases", + "url": "https://pypi.org/project/strands-agents/", + "date": "2026-07-09", + "value": "Rapid post-1.0 cadence continues: Python SDK 1.46.0 (2026-07-08) and TypeScript 1.8.0 (2026-07-08)" } ], "methodology": "Production readiness assessment of API stability and documented production usage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -463,8 +475,10 @@ "strands-agents-tools library", "A2A agent interoperability" ], - "github_stars": "10000+", + "github_stars": "6500+ (harness-sdk monorepo, July 2026)", + "github_repo": "https://github.com/strands-agents/harness-sdk (renamed from sdk-python, June 2026)", "first_release": "2025 (open-sourced May 2025)", + "latest_version": "Python SDK 1.46.0 / TypeScript SDK 1.8.0 (2026-07-08)", "pricing": "Free (Apache 2.0) - Costs only from LLM usage and optional AWS services", "python_requirement": "Python >=3.10", "adoption": "14M+ downloads by Feb 2026; used internally by Amazon Q Developer and multiple AWS teams" diff --git a/data/agents/superagi.json b/data/agents/superagi.json index 80efbf2..85657a2 100644 --- a/data/agents/superagi.json +++ b/data/agents/superagi.json @@ -4,9 +4,9 @@ "name": "SuperAGI", "provider": "Community", "version": "0.x", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source autonomous AI agent framework for running and managing multiple AI agents concurrently. Features GUI-based management, tool marketplace, and agent templates for building production-ready autonomous agents.", + "description": "UNMAINTAINED: the SuperAGI repository is dormant — the last commit was January 2025 (an IDOR security fix) and the last release was v0.0.14 in January 2024; it is not recommended for new deployments. Formerly an open-source autonomous AI agent framework for running and managing multiple AI agents concurrently, with GUI-based management, a tool marketplace, and agent templates.", "website": "https://superagi.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Autonomous task testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_agent_orchestration": { "score": 78, @@ -38,7 +38,7 @@ } ], "methodology": "Multi-agent coordination testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_integration": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Tool integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "planning_capability": { "score": 74, @@ -66,7 +66,7 @@ } ], "methodology": "Planning effectiveness testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_management": { "score": 72, @@ -80,7 +80,7 @@ } ], "methodology": "Memory system evaluation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "2-15s per iteration", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "self_hosting": { "score": 85, @@ -127,7 +127,7 @@ } ], "methodology": "Deployment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "authentication": { "score": 72, @@ -141,7 +141,7 @@ } ], "methodology": "Authentication assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 88, @@ -150,12 +150,12 @@ { "source": "GitHub", "url": "https://github.com/TransformerOptimus/SuperAGI", - "date": "2024-10-20", - "value": "MIT license, 15k+ stars, open source community" + "date": "2026-07-09", + "value": "MIT license, 17.6k stars, source remains fully available but development is dormant" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_security": { "score": 68, @@ -169,7 +169,7 @@ } ], "methodology": "API security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 77, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_deployment": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Deployment options assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "llm_data_sharing": { "score": 72, @@ -230,7 +230,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "agent_logs": { "score": 76, @@ -244,12 +244,12 @@ } ], "methodology": "Data storage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "trust_transparency": { - "overall_score": 80, + "overall_score": 74, "criteria": { "documentation_quality": { "score": 78, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gui_interface": { "score": 85, @@ -277,7 +277,7 @@ } ], "methodology": "UI/UX assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 90, @@ -286,12 +286,12 @@ { "source": "GitHub", "url": "https://github.com/TransformerOptimus/SuperAGI", - "date": "2024-10-20", - "value": "MIT license, 15k+ stars, active community" + "date": "2026-07-09", + "value": "MIT license, 17.6k stars; code transparent but no active development since January 2025" } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "agent_traceability": { "score": 76, @@ -305,26 +305,33 @@ } ], "methodology": "Traceability features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { - "score": 75, - "confidence": "medium", + "score": 45, + "confidence": "high", "evidence": [ { "source": "Community", "url": "https://github.com/TransformerOptimus/SuperAGI/discussions", "date": "2024-10-15", "value": "Growing community, Discord support" + }, + { + "source": "GitHub Activity", + "url": "https://github.com/TransformerOptimus/SuperAGI/issues/1431", + "date": "2026-07-09", + "value": "Maintainer activity has ceased (no commits since 2025-01-22); community itself flagged the project as inactive (issue #1431) and open issues go unanswered" } ], "methodology": "Community engagement analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09", + "notes": "Score reduced from 75: maintainers unresponsive and development dormant since January 2025" } } }, "operational_excellence": { - "overall_score": 72, + "overall_score": 67, "criteria": { "ease_of_setup": { "score": 75, @@ -338,7 +345,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 68, @@ -352,7 +359,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 90, @@ -366,7 +373,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 70, @@ -380,7 +387,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tool_marketplace": { "score": 82, @@ -394,28 +401,35 @@ } ], "methodology": "Tool ecosystem assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { - "score": 67, - "confidence": "medium", + "score": 35, + "confidence": "high", "evidence": [ { "source": "Maturity", "url": "https://github.com/TransformerOptimus/SuperAGI", "date": "2024-10-01", "value": "Early stage project, production use requires hardening" + }, + { + "source": "GitHub Repository Status", + "url": "https://github.com/TransformerOptimus/SuperAGI/commits", + "date": "2026-07-09", + "value": "Repository dormant: last commit 2025-01-22 (IDOR security fix), last release v0.0.14 (2024-01-16); no maintenance for ~18 months, unresolved issue backlog" } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09", + "notes": "Score reduced from 67: project is effectively unmaintained (no commits since January 2025, no releases since January 2024), so unpatched dependencies and abandoned issue backlog make production use inadvisable" } } } }, "strengths": [ "GUI-based agent management for easy monitoring and control", - "Open source (MIT) with growing community (15k+ stars)", + "Open source (MIT) with historically large community (17k+ stars)", "Tool marketplace with 40+ pre-built tools and extensibility", "Multi-agent orchestration for concurrent autonomous tasks", "Self-hosted deployment for data privacy and control", @@ -427,7 +441,8 @@ "Can be expensive due to multiple LLM calls in autonomous loops", "Autonomous agent reliability can be unpredictable", "Requires technical knowledge for setup and customization", - "Limited documentation compared to more mature frameworks" + "Limited documentation compared to more mature frameworks", + "Unmaintained (2026-07): no commits since January 2025 and no releases since v0.0.14 (January 2024); security patches and dependency updates are not being shipped" ], "metadata": { "license": "MIT", @@ -447,8 +462,9 @@ "API integrations" ], "pricing_model": "Free open source", - "github_stars": "16600+", + "github_stars": "17600+", "first_release": "2023", + "status": "Dormant/unmaintained: last commit 2025-01-22, last release v0.0.14 (2024-01-16); repo not archived but inactive", "database": "PostgreSQL", "agent_features": [ "Goal-oriented", @@ -508,6 +524,7 @@ "tags": [ "autonomous", "self-hosted", - "open-source" + "open-source", + "unmaintained" ] } diff --git a/data/agents/swarm.json b/data/agents/swarm.json index 273ca8b..9bace3e 100644 --- a/data/agents/swarm.json +++ b/data/agents/swarm.json @@ -4,9 +4,9 @@ "name": "OpenAI Swarm", "provider": "OpenAI", "version": "Experimental (Deprecated)", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DEPRECATED: archived since 2025-03-11; the README redirects users to the OpenAI Agents SDK as the production successor. Swarm was an experimental educational framework from OpenAI for multi-agent orchestration, demonstrating agent coordination and handoff patterns with simple Python primitives. Not maintained; do not use for new projects.", + "description": "DEPRECATED: marked deprecated since 2025-03-11; the README redirects users to the OpenAI Agents SDK as the production successor (repo remains public and unmaintained, not GitHub-archived). Swarm was an experimental educational framework from OpenAI for multi-agent orchestration, demonstrating agent coordination and handoff patterns with simple Python primitives. Not maintained; do not use for new projects.", "website": "https://github.com/openai/swarm", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Orchestration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "agent_handoffs": { "score": 80, @@ -38,7 +38,7 @@ } ], "methodology": "Handoff testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "simplicity": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Code complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "context_management": { "score": 72, @@ -66,7 +66,7 @@ } ], "methodology": "Context handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "experimental_status": { "score": 60, @@ -80,7 +80,7 @@ } ], "methodology": "Status assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "Variable (OpenAI API dependent)", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Dependency analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "openai_trust": { "score": 88, @@ -127,7 +127,7 @@ } ], "methodology": "Source trust assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "experimental_risks": { "score": 55, @@ -141,7 +141,7 @@ } ], "methodology": "Security maturity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 90, @@ -155,7 +155,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_security": { "score": 68, @@ -169,7 +169,7 @@ } ], "methodology": "API security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "openai_data_sharing": { "score": 70, @@ -202,7 +202,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "no_telemetry": { "score": 85, @@ -216,7 +216,7 @@ } ], "methodology": "Telemetry assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_considerations": { "score": 72, @@ -230,7 +230,7 @@ } ], "methodology": "Compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_control": { "score": 68, @@ -244,7 +244,7 @@ } ], "methodology": "Data control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "code_clarity": { "score": 90, @@ -277,7 +277,7 @@ } ], "methodology": "Code quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "openai_backing": { "score": 88, @@ -291,7 +291,7 @@ } ], "methodology": "Source trust assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source": { "score": 90, @@ -305,7 +305,7 @@ } ], "methodology": "Open source assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "educational_purpose": { "score": 75, @@ -319,7 +319,7 @@ } ], "methodology": "Purpose assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 25, @@ -353,13 +353,13 @@ { "source": "GitHub Repository Status", "url": "https://github.com/openai/swarm", - "date": "2026-06-10", - "value": "Repository deprecated/archived since 2025-03-11; README redirects to the OpenAI Agents SDK as the production successor" + "date": "2026-07-09", + "value": "Repository deprecated since 2025-03-11; README redirects to the OpenAI Agents SDK as the production successor; repo public but unmaintained (not GitHub-archived), ~21,800 stars" } ], "methodology": "Production readiness assessment", - "last_verified": "2026-06-10", - "notes": "Score reduced: project archived since 2025-03-11 with no further maintenance" + "last_verified": "2026-07-09", + "notes": "Score reduced: project deprecated since 2025-03-11 with no further maintenance" }, "scalability": { "score": 65, @@ -373,7 +373,7 @@ } ], "methodology": "Scalability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 88, @@ -387,7 +387,7 @@ } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 58, @@ -401,7 +401,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "learning_value": { "score": 92, @@ -415,7 +415,7 @@ } ], "methodology": "Educational value assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -426,7 +426,7 @@ "Clean code that demonstrates best practices", "MIT licensed with minimal dependencies", "Excellent for learning agent orchestration concepts", - "13k+ GitHub stars showing community interest" + "21k+ GitHub stars showing lasting educational interest" ], "limitations": [ "Explicitly experimental, not for production use", @@ -435,7 +435,7 @@ "No built-in monitoring, error handling, or scaling features", "Minimal documentation beyond examples", "Not actively maintained for production use cases", - "Deprecated/archived since 2025-03-11; README directs users to the OpenAI Agents SDK" + "Deprecated since 2025-03-11; README directs users to the OpenAI Agents SDK" ], "metadata": { "license": "MIT", @@ -452,7 +452,7 @@ "Agent handoffs" ], "pricing_model": "Free open source (OpenAI API costs apply)", - "github_stars": "13000+", + "github_stars": "21700+", "first_release": "2024", "status": "Experimental / Educational - Replaced by OpenAI Agents SDK", "deprecation_notice": "OpenAI recommends migrating to the Agents SDK for production use. Swarm was experimental only and is not officially supported.", diff --git a/data/agents/zapier-ai.json b/data/agents/zapier-ai.json index 64e5cc8..4e2ddb4 100644 --- a/data/agents/zapier-ai.json +++ b/data/agents/zapier-ai.json @@ -4,9 +4,9 @@ "name": "Zapier AI Actions", "provider": "Zapier", "version": "Current", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "No-code automation platform with AI actions for connecting AI models to 6000+ apps. Enables building intelligent automations and AI-powered workflows without coding through visual interface and natural language.", + "description": "No-code automation platform with AI actions for connecting AI models to 8,000+ apps. Enables building intelligent automations and AI-powered workflows without coding through visual interface and natural language. Now includes Zapier Agents (agents.zapier.com) for autonomous AI agents, Chatbots, and Zapier MCP for connecting external AI assistants to its app ecosystem.", "website": "https://zapier.com/ai", "trust_vector": { "performance_reliability": { @@ -24,21 +24,21 @@ } ], "methodology": "Workflow reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ai_action_performance": { "score": 85, "confidence": "high", "evidence": [ { - "source": "AI Actions", - "url": "https://actions.zapier.com/", - "date": "2024-10-15", - "value": "AI actions integrate with GPT and other LLMs" + "source": "AI Actions / Zapier Agents", + "url": "https://zapier.com/agents", + "date": "2026-07-09", + "value": "AI Actions integrate with LLMs; Zapier Agents product provides autonomous agents built into Zap workflows across all plans (with paid add-ons)" } ], "methodology": "AI integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "integration_reliability": { "score": 90, @@ -47,12 +47,12 @@ { "source": "App Integrations", "url": "https://zapier.com/apps", - "date": "2024-10-01", - "value": "6000+ app integrations with high uptime" + "date": "2026-07-09", + "value": "8,000+ app integrations with high uptime (9,000+ reachable via Zapier MCP/SDK)" } ], "methodology": "Integration testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "chatbots": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Chatbot capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 83, @@ -80,7 +80,7 @@ } ], "methodology": "Error recovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "latency": { "value": "1-30s average", @@ -94,7 +94,7 @@ } ], "methodology": "Performance monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "authentication": { "score": 89, @@ -127,7 +127,7 @@ } ], "methodology": "Authentication testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_storage": { "score": 88, @@ -141,7 +141,7 @@ } ], "methodology": "Credential security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance": { "score": 90, @@ -155,7 +155,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "app_permissions": { "score": 82, @@ -169,7 +169,7 @@ } ], "methodology": "Permission model assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 89, @@ -202,7 +202,7 @@ } ], "methodology": "Compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "hipaa_compliance": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Healthcare compliance review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_processing": { "score": 83, @@ -230,7 +230,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_deletion": { "score": 86, @@ -244,7 +244,7 @@ } ], "methodology": "Data rights assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "visual_editor": { "score": 92, @@ -277,7 +277,7 @@ } ], "methodology": "UI/UX assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "task_history": { "score": 84, @@ -291,7 +291,7 @@ } ], "methodology": "Traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "template_library": { "score": 87, @@ -305,7 +305,7 @@ } ], "methodology": "Template availability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "support_quality": { "score": 80, @@ -319,7 +319,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -338,7 +338,7 @@ } ], "methodology": "Usability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scalability": { "score": 89, @@ -352,7 +352,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 79, @@ -361,12 +361,12 @@ { "source": "Pricing", "url": "https://zapier.com/pricing", - "date": "2024-10-01", - "value": "Task-based pricing, free tier + paid from $19.99/month" + "date": "2026-07-09", + "value": "Task-based pricing: Free (100 tasks/month), Professional from $19.99/month (annual billing, 750 tasks), Team from $69/month, Enterprise custom; AI add-ons (Agents, Chatbots) priced separately" } ], "methodology": "Pricing model analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitoring": { "score": 85, @@ -380,7 +380,7 @@ } ], "methodology": "Monitoring features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "integration_ecosystem": { "score": 98, @@ -389,12 +389,12 @@ { "source": "Apps", "url": "https://zapier.com/apps", - "date": "2024-10-01", - "value": "6000+ app integrations, industry leader" + "date": "2026-07-09", + "value": "8,000+ app integrations, industry leader" } ], "methodology": "Integration ecosystem assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "uptime": { "score": 88, @@ -408,17 +408,17 @@ } ], "methodology": "Uptime monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } }, "strengths": [ - "Industry-leading 6000+ app integrations, largest ecosystem", + "Industry-leading 8,000+ app integrations, largest ecosystem", "Exceptional no-code ease of use for non-technical users", "Strong enterprise security and compliance (SOC 2, ISO 27001, HIPAA)", "Reliable infrastructure with high uptime and automatic retries", - "AI Actions for integrating GPTs and chatbots with apps", + "AI Actions, Zapier Agents, Chatbots, and Zapier MCP for AI-app integration", "Extensive template library and documentation" ], "limitations": [ @@ -442,25 +442,27 @@ ], "deployment_type": "Managed cloud service", "tool_support": [ - "6000+ app integrations", + "8,000+ app integrations", "Webhooks", - "API requests" + "API requests", + "Zapier MCP" ], - "pricing_model": "Free tier + paid plans ($19.99-$69+/month)", + "pricing_model": "Free tier + paid plans ($19.99-$69+/month, annual billing)", "tasks_included": "100 to unlimited per month", "founded": "2011", "users": "6+ million", "ai_features": [ "AI Actions", + "Zapier Agents", "Chatbots", - "ChatGPT plugin" + "Zapier MCP" ], - "pricing": "Free: 400 activities/month; Pro: $33.33/month for 1,500 activities; Chatbots from $20/month; Platform from $19.99/month" + "pricing": "Free: 100 tasks/month; Professional from $19.99/month (annual, 750 tasks); Team from $69/month; Enterprise custom; Zapier Agents and Chatbots available as add-ons" }, "use_case_ratings": { "customer-support": { "overall": 90, - "notes": "Excellent for automating support across 6000+ tools" + "notes": "Excellent for automating support across 8,000+ tools" }, "code-generation": { "overall": 70, @@ -501,7 +503,7 @@ }, "best_for": [ "Business users automating workflows with AI actions", - "Teams needing no-code AI integration with 6000+ apps", + "Teams needing no-code AI integration with 8,000+ apps", "Organizations wanting quick AI-powered automation", "Non-technical users adding AI to existing Zapier workflows" ], diff --git a/data/mcps/mcp-server-apify.json b/data/mcps/mcp-server-apify.json index 41c4b44..f82ffce 100644 --- a/data/mcps/mcp-server-apify.json +++ b/data/mcps/mcp-server-apify.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "Apify MCP Server", "provider": "Apify", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official Apify MCP server (renamed from actors-mcp-server) exposing thousands of Apify Store Actors to AI agents — scrapers for social media, maps, e-commerce, and search. Agents can dynamically discover and run Actors via the @apify/actors-mcp-server package or the hosted OAuth endpoint at mcp.apify.com.", + "description": "Official Apify MCP server (renamed from actors-mcp-server) exposing thousands of Apify Store Actors - scrapers for social media, maps, e-commerce, and search - via the @apify/actors-mcp-server package (v0.11.x) or the hosted endpoint at mcp.apify.com (OAuth or Bearer API token, Streamable HTTP only; legacy SSE removed April 2026). The hosted server adds output-schema inference, agentic payments (x402 USDC on Base, Skyfire), and June 2026 MCP connectors for login-required apps.", "website": "https://github.com/apify/apify-mcp-server", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Platform stability and maturity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "actor_execution_success": { "score": 78, @@ -38,7 +38,7 @@ } ], "methodology": "Run success sampling across official and community Actors", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "dynamic_tool_discovery": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Tool discovery relevance testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Rate limiting behavior review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Failure mode and recovery testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -92,14 +92,14 @@ "confidence": "high", "evidence": [ { - "source": "Apify MCP Documentation", - "url": "https://mcp.apify.com/", - "date": "2026-06-10", - "value": "Hosted server at mcp.apify.com uses OAuth; local stdio mode uses an Apify API token" + "source": "Apify MCP Server Repository", + "url": "https://github.com/apify/apify-mcp-server", + "date": "2026-07-09", + "value": "Hosted server at mcp.apify.com uses OAuth or an Authorization Bearer Apify API token over Streamable HTTP (legacy SSE endpoint removed April 2026); local stdio mode uses an Apify API token; agentic payments via x402 (USDC on Base) or Skyfire allow runs without a token" } ], "methodology": "Authentication mechanism review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "dynamic_capability_risk": { "score": 55, @@ -113,7 +113,7 @@ } ], "methodology": "Runtime capability mutation threat modeling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "actor_supply_chain_risk": { "score": 58, @@ -127,7 +127,7 @@ } ], "methodology": "Supply chain analysis of community-published Actors", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_handling": { "score": 78, @@ -141,7 +141,7 @@ } ], "methodology": "Token scope and exposure analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sandboxed_execution": { "score": 80, @@ -155,7 +155,7 @@ } ], "methodology": "Execution isolation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of Actor inputs and outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 62, @@ -188,7 +188,7 @@ } ], "methodology": "PII handling assessment of common Actor outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 62, @@ -202,7 +202,7 @@ } ], "methodology": "Data sharing and policy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_posture": { "score": 75, @@ -216,7 +216,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 84, @@ -249,7 +249,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -258,12 +258,12 @@ { "source": "Apify MCP Server Repository", "url": "https://github.com/apify/apify-mcp-server", - "date": "2026-06-10", - "value": "MIT-licensed open source server with 1,319 GitHub stars; many Store Actors also publish source, though not all" + "date": "2026-07-09", + "value": "MIT-licensed open source server with 1,755 GitHub stars; many Store Actors also publish source, though not all" } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_coverage_clarity": { "score": 80, @@ -272,12 +272,12 @@ { "source": "Apify MCP Server Repository", "url": "https://github.com/apify/apify-mcp-server", - "date": "2026-06-10", - "value": "Helper tools are documented, but the effective tool set is dynamic — it depends on which Actors are loaded at runtime" + "date": "2026-07-09", + "value": "Helper tools are documented (search-actors, fetch-actor-details, call-actor, dataset/key-value access, run management, docs search), but the effective tool set is dynamic — it depends on which Actors are loaded at runtime; hosted server adds output-schema inference for structured results" } ], "methodology": "Tool surface documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 80, @@ -310,7 +310,7 @@ } ], "methodology": "Run latency characterization", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 84, @@ -324,7 +324,7 @@ } ], "methodology": "Uptime and incident history analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 92, @@ -338,7 +338,7 @@ } ], "methodology": "Capability breadth assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 79, @@ -347,12 +347,12 @@ { "source": "GitHub Repository Metrics", "url": "https://github.com/apify/apify-mcp-server", - "date": "2026-06-10", - "value": "1,319 GitHub stars with active maintenance by Apify and a large existing Actor developer ecosystem" + "date": "2026-07-09", + "value": "1,755 GitHub stars (up from 1,319 in June 2026) with active maintenance by Apify (v0.11.5 released July 2026) and a large existing Actor developer ecosystem" } ], "methodology": "Community activity and adoption analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -361,7 +361,8 @@ "Access to thousands of ready-made Apify Store Actors for social media, maps, e-commerce, and search scraping", "Dynamic tool discovery lets agents find and add the right Actor at runtime", "Actors run in isolated containers in Apify's cloud, not on the user's machine", - "Hosted OAuth endpoint (mcp.apify.com) and simple local npx setup", + "Hosted OAuth endpoint (mcp.apify.com, Streamable HTTP) and simple local npx setup", + "Hosted server infers output schemas for structured Actor results", "Full run auditability via the Apify console with logs, inputs, and cost tracking", "MIT-licensed open source server actively maintained by Apify" ], @@ -369,7 +370,8 @@ "Actors are community-published code of varying quality and provenance", "Dynamically added tools mutate the agent's capability surface after initial approval", "Scraped output frequently contains personal data with GDPR implications and no built-in PII filtering", - "Actor runs consume paid platform credits that an agent can spend autonomously", + "Actor runs consume paid platform credits that an agent can spend autonomously; agentic payment rails (x402, Skyfire) extend this to direct token-less spending", + "MCP connectors (June 2026) let Actors act on login-required apps (Notion, Slack, GitHub), widening the blast radius beyond scraping", "Community Actors can silently break when target websites change", "Scraped content is untrusted input that may carry prompt injection payloads" ], @@ -383,15 +385,16 @@ "TypeScript" ], "github_repo": "https://github.com/apify/apify-mcp-server", - "github_stars": 1319, + "github_stars": 1755, "package_name": "@apify/actors-mcp-server", + "package_version": "0.11.5", "previous_name": "actors-mcp-server", "api_dependency": "Apify platform API", - "authentication": "OAuth (hosted) or Apify API token (stdio)", + "authentication": "OAuth or Bearer API token (hosted); Apify API token (stdio); agentic payments via x402 (USDC on Base) or Skyfire", "maintained_by": "Apify", "transport_types": [ "stdio", - "remote (hosted)" + "streamable-http (hosted; legacy SSE removed 2026-04)" ], "installation_methods": [ "npm", diff --git a/data/mcps/mcp-server-atlassian.json b/data/mcps/mcp-server-atlassian.json index de05c6d..39931a2 100644 --- a/data/mcps/mcp-server-atlassian.json +++ b/data/mcps/mcp-server-atlassian.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Atlassian Server", "provider": "Atlassian", - "version": "1.0.0", - "last_evaluated": "2025-11-09", + "version": "hosted remote (GA 2026-02)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official Atlassian MCP server for Jira and Confluence integration. Enables AI models to interact with project management, issue tracking, and knowledge base operations. Essential for AI-powered project management, documentation, and team collaboration workflows.", + "description": "Official Atlassian (Rovo) MCP server, generally available since February 2026 as a hosted remote at https://mcp.atlassian.com/v1/mcp (OAuth 2.1 via /v1/mcp/authv2, or API-token auth for machine-to-machine use). Connects Jira, Confluence, Jira Service Management, Bitbucket, and Compass to AI models for project management, issue tracking, and knowledge base operations. The legacy /v1/sse endpoint is unsupported after 2026-06-30.", "website": "https://github.com/atlassian/atlassian-mcp-server", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "API stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "issue_operation_success": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 78, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Search result quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -96,10 +96,16 @@ "url": "https://developer.atlassian.com/cloud/jira/platform/security-overview/", "date": "2025-11-16", "value": "Uses Atlassian API tokens or OAuth 2.0 for secure authentication" + }, + { + "source": "Atlassian Support - Configuring OAuth 2.1 (Rovo MCP Server)", + "url": "https://support.atlassian.com/atlassian-rovo-mcp-server/docs/configuring-oauth-2-1/", + "date": "2026-07-09", + "value": "Hosted Rovo MCP server uses browser-based OAuth 2.1 (endpoint https://mcp.atlassian.com/v1/mcp/authv2); API-token auth added for machine-to-machine use without an interactive consent screen" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 68, @@ -113,7 +119,7 @@ } ], "methodology": "Token security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "permission_scope_control": { "score": 75, @@ -127,7 +133,7 @@ } ], "methodology": "Permission scope testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_modification_risk": { "score": 70, @@ -141,7 +147,7 @@ } ], "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 85, @@ -155,7 +161,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cross_project_access": { "score": 72, @@ -169,7 +175,7 @@ } ], "methodology": "Access boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +194,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sensitive_issue_protection": { "score": 65, @@ -202,7 +208,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "team_member_privacy": { "score": 72, @@ -216,7 +222,7 @@ } ], "methodology": "User privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 72, @@ -230,7 +236,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "attachment_handling": { "score": 75, @@ -244,7 +250,7 @@ } ], "methodology": "File handling assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +269,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 87, @@ -277,7 +283,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 90, @@ -291,7 +297,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 75, @@ -305,7 +311,7 @@ } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -321,10 +327,16 @@ "url": "https://github.com/atlassian/atlassian-mcp-server", "date": "2025-11-16", "value": "Requires Atlassian API token and site configuration" + }, + { + "source": "Atlassian - Remote MCP Server", + "url": "https://www.atlassian.com/platform/remote-mcp-server", + "date": "2026-07-09", + "value": "Hosted remote requires no install: connect to https://mcp.atlassian.com/v1/mcp and complete browser OAuth 2.1 (or configure an API token)" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_performance": { "score": 80, @@ -338,7 +350,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -352,7 +364,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 82, @@ -363,10 +375,16 @@ "url": "https://github.com/atlassian/atlassian-mcp-server", "date": "2025-11-16", "value": "Covers Jira issues, JQL search, Confluence pages, and spaces" + }, + { + "source": "Atlassian MCP Server (official repo)", + "url": "https://github.com/atlassian/atlassian-mcp-server", + "date": "2026-07-09", + "value": "Coverage expanded beyond Jira/Confluence to Jira Service Management, Bitbucket, and Compass" } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 78, @@ -377,10 +395,16 @@ "url": "https://github.com/atlassian/atlassian-mcp-server/discussions", "date": "2025-11-16", "value": "Growing adoption among Atlassian users" + }, + { + "source": "Atlassian Community - Rovo MCP server GA announcement", + "url": "https://community.atlassian.com/forums/Atlassian-Remote-MCP-Server/Announcing-GA-of-the-Atlassian-Rovo-MCP-server/ba-p/3186690", + "date": "2026-07-09", + "value": "Rovo MCP server generally available (GA February 2026) with an active dedicated community forum" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -435,7 +459,8 @@ ], "strengths": [ "Official Atlassian implementation with deep Jira and Confluence expertise", - "Comprehensive project management and documentation capabilities", + "Hosted remote (Rovo MCP Server) generally available since 2026-02 with OAuth 2.1 or API-token auth", + "Coverage spans Jira, Confluence, Jira Service Management, Bitbucket, and Compass for project management and documentation", "Built on reliable Atlassian Cloud APIs with strong SLAs", "Full operation auditability through Atlassian audit logs", "Open source with Atlassian support", @@ -447,7 +472,8 @@ "No built-in filtering for sensitive or confidential content", "May expose team member names, emails, and activity patterns", "Subject to Atlassian API rate limits", - "Attachments and file content may be transmitted to LLM provider" + "Attachments and file content may be transmitted to LLM provider", + "Legacy SSE endpoint (https://mcp.atlassian.com/v1/sse) unsupported after 2026-06-30; clients must migrate to /v1/mcp" ], "metadata": { "license": "MIT", @@ -460,15 +486,24 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/atlassian/atlassian-mcp-server", - "api_dependency": "Jira REST API, Confluence REST API", - "authentication": "Atlassian API Token or OAuth 2.0", + "api_dependency": "Jira REST API, Confluence REST API, Jira Service Management, Bitbucket, Compass", + "authentication": "OAuth 2.1 (browser flow) or Atlassian API token (machine-to-machine)", + "remote_endpoint": "https://mcp.atlassian.com/v1/mcp (OAuth: https://mcp.atlassian.com/v1/mcp/authv2)", + "remote_ga_date": "2026-02", + "deprecated_endpoint": "https://mcp.atlassian.com/v1/sse (unsupported after 2026-06-30)", "first_release": "2024-11", - "maintained_by": "Atlassian" + "maintained_by": "Atlassian", + "status": "Active - Atlassian Rovo MCP Server GA since 2026-02", + "transport_types": [ + "streamable-http (hosted remote)" + ] }, "tags": [ "project-management", "jira", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "remote-server" ] } diff --git a/data/mcps/mcp-server-aws.json b/data/mcps/mcp-server-aws.json index 3aa2e89..47d197d 100644 --- a/data/mcps/mcp-server-aws.json +++ b/data/mcps/mcp-server-aws.json @@ -2,12 +2,12 @@ "id": "mcp-server-aws", "type": "mcp", "name": "MCP AWS Server", - "provider": "Community", - "version": "1.0.0", - "last_evaluated": "2025-11-08", + "provider": "AWS", + "version": "2026.5 (managed server GA)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "MCP server enabling AI models to interact with AWS cloud services including S3, EC2, Lambda, and more. Supports infrastructure management, resource provisioning, and cloud automation through the Model Context Protocol. Extremely powerful but poses critical security risks requiring strict controls.", - "website": "https://aws.amazon.com/sdk-for-javascript/", + "description": "AWS's official MCP offering for AWS cloud services. Two forms: the managed AWS MCP Server (GA 2026-05-06), a fully managed remote exposing 15,000+ AWS API operations plus documentation search and a sandboxed run_script tool, accessed through the open-source MCP Proxy for AWS which bridges IAM SigV4 credentials to MCP's OAuth model; and the awslabs/mcp suite of open-source, service-scoped servers (DynamoDB, CDK, pricing, etc.). Powerful but requires strict IAM controls.", + "website": "https://github.com/awslabs/mcp", "trust_vector": { "performance_reliability": { "overall_score": 85, @@ -24,7 +24,7 @@ } ], "methodology": "API uptime and reliability analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Operation success testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "multi_service_integration": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Service integration testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "response_time": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "API latency testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "error_handling": { "score": 84, @@ -80,40 +80,40 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 64, + "overall_score": 67, "criteria": { "iam_security": { - "score": 78, + "score": 82, "confidence": "high", "evidence": [ { - "source": "AWS IAM", - "url": "https://aws.amazon.com/iam/", - "date": "2025-11-16", - "value": "Granular IAM permissions available but AI can use all granted permissions" + "source": "AWS MCP Server GA Announcement", + "url": "https://aws.amazon.com/blogs/aws/the-aws-mcp-server-is-now-generally-available/", + "date": "2026-07-09", + "value": "Managed server authenticates with IAM credentials via SigV4; supports read-only restriction through IAM policies or Service Control Policies and separation between human and agent permissions" } ], - "methodology": "IAM security review", - "last_verified": "2025-11-08" + "methodology": "IAM security review; score raised from 78 as the GA managed server adds documented agent-specific IAM guidance, read-only policy patterns, and SCP-based governance", + "last_verified": "2026-07-09" }, "credential_exposure_risk": { - "score": 60, + "score": 68, "confidence": "high", "evidence": [ { - "source": "AWS Credentials", - "url": "https://docs.aws.amazon.com/sdk-for-javascript/v3/developer-guide/setting-credentials.html", - "date": "2025-11-16", - "value": "Access keys stored locally but AI can perform any action within permissions" + "source": "AWS MCP Server GA Announcement", + "url": "https://aws.amazon.com/blogs/aws/the-aws-mcp-server-is-now-generally-available/", + "date": "2026-07-09", + "value": "The open-source MCP Proxy for AWS bridges IAM (SigV4) to MCP's OAuth model, so short-lived IAM role credentials can be used instead of static access keys; self-hosted awslabs servers still commonly rely on locally configured credentials" } ], - "methodology": "Credential security analysis", - "last_verified": "2025-11-08" + "methodology": "Credential security analysis; modest raise from 60 since the managed server supports role-based short-lived credentials rather than requiring static keys", + "last_verified": "2026-07-09" }, "resource_modification_risk": { "score": 52, @@ -127,7 +127,7 @@ } ], "methodology": "Resource control risk assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "cost_control_risk": { "score": 55, @@ -141,7 +141,7 @@ } ], "methodology": "Cost control assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 85, @@ -152,10 +152,16 @@ "url": "https://aws.amazon.com/cloudtrail/", "date": "2025-11-16", "value": "Comprehensive audit logging via CloudTrail for all API actions" + }, + { + "source": "AWS MCP Server GA Announcement", + "url": "https://aws.amazon.com/blogs/aws/the-aws-mcp-server-is-now-generally-available/", + "date": "2026-07-09", + "value": "Managed server additionally publishes CloudWatch metrics under the AWS-MCP namespace and logs agent actions in CloudTrail for compliance" } ], "methodology": "Audit capabilities review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_exfiltration_risk": { "score": 58, @@ -169,7 +175,7 @@ } ], "methodology": "Data exposure analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -188,7 +194,7 @@ } ], "methodology": "Data exposure analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 60, @@ -202,7 +208,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "vpc_isolation": { "score": 75, @@ -216,7 +222,7 @@ } ], "methodology": "Network isolation assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_readiness": { "score": 68, @@ -230,7 +236,7 @@ } ], "methodology": "Compliance framework review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "encryption_support": { "score": 72, @@ -244,12 +250,12 @@ } ], "methodology": "Encryption capabilities review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, "trust_transparency": { - "overall_score": 80, + "overall_score": 85, "criteria": { "documentation_quality": { "score": 88, @@ -263,7 +269,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "action_visibility": { "score": 85, @@ -277,68 +283,68 @@ } ], "methodology": "Action traceability assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "mcp_implementation": { - "score": 72, - "confidence": "medium", + "score": 84, + "confidence": "high", "evidence": [ { - "source": "Community Implementation", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Community-maintained with variable documentation" + "source": "awslabs/mcp Repository", + "url": "https://github.com/awslabs/mcp", + "date": "2026-07-09", + "value": "First-party AWS implementation: open-source awslabs/mcp suite plus the managed AWS MCP Server (GA 2026-05-06) with published documentation site at awslabs.github.io/mcp" } ], - "methodology": "Implementation documentation review", - "last_verified": "2025-11-08" + "methodology": "Implementation documentation review; raised from 72 as the implementation is now officially maintained by AWS with a dedicated docs site, replacing the previous community-maintained status", + "last_verified": "2026-07-09" }, "security_guidance": { - "score": 75, - "confidence": "medium", + "score": 82, + "confidence": "high", "evidence": [ { - "source": "AWS Security Best Practices", - "url": "https://aws.amazon.com/security/", - "date": "2025-11-16", - "value": "AWS provides security guidance but specific MCP guidance limited" + "source": "AWS MCP Server GA Announcement", + "url": "https://aws.amazon.com/blogs/aws/the-aws-mcp-server-is-now-generally-available/", + "date": "2026-07-09", + "value": "AWS now publishes MCP-specific security guidance: IAM/SCP read-only patterns, separation of human vs agent permissions, and per-server security documentation in awslabs/mcp" } ], - "methodology": "Security documentation review", - "last_verified": "2025-11-08" + "methodology": "Security documentation review; raised from 75 due to official MCP-specific IAM governance guidance shipped with GA", + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 77, + "overall_score": 79, "criteria": { "ease_of_setup": { - "score": 70, + "score": 74, "confidence": "medium", "evidence": [ { - "source": "AWS SDK Setup", - "url": "https://docs.aws.amazon.com/sdk-for-javascript/v3/developer-guide/getting-started.html", - "date": "2025-11-16", - "value": "Requires AWS account, IAM configuration, and credential management" + "source": "AWS MCP Server GA Announcement", + "url": "https://aws.amazon.com/blogs/aws/the-aws-mcp-server-is-now-generally-available/", + "date": "2026-07-09", + "value": "Managed server requires installing the local MCP Proxy for AWS plus IAM configuration; awslabs servers install via uvx/pip; still requires AWS account and credential management" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_coverage": { - "score": 85, + "score": 92, "confidence": "high", "evidence": [ { - "source": "AWS Services", - "url": "https://aws.amazon.com/products/", - "date": "2025-11-16", - "value": "Supports major AWS services (S3, EC2, Lambda, RDS, etc.)" + "source": "AWS MCP Server GA Announcement", + "url": "https://aws.amazon.com/blogs/aws/the-aws-mcp-server-is-now-generally-available/", + "date": "2026-07-09", + "value": "Managed server's call_aws tool executes any of 15,000+ AWS API operations across all services, plus documentation search/read and sandboxed server-side Python via run_script" } ], - "methodology": "API coverage assessment", - "last_verified": "2025-11-08" + "methodology": "API coverage assessment; raised from 85 as the GA managed server covers essentially the full AWS API surface", + "last_verified": "2026-07-09" }, "reliability": { "score": 82, @@ -352,7 +358,7 @@ } ], "methodology": "Uptime analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "cost_predictability": { "score": 65, @@ -366,21 +372,21 @@ } ], "methodology": "Cost predictability analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "community_support": { - "score": 75, - "confidence": "medium", + "score": 82, + "confidence": "high", "evidence": [ { - "source": "Community", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Growing community for MCP AWS integration" + "source": "awslabs/mcp Repository", + "url": "https://github.com/awslabs/mcp", + "date": "2026-07-09", + "value": "Actively maintained first-party open-source suite with a large contributor base; managed server backed by AWS support channels" } ], - "methodology": "Community activity analysis", - "last_verified": "2025-11-08" + "methodology": "Community activity analysis; raised from 75 reflecting first-party AWS maintenance", + "last_verified": "2026-07-09" } } } @@ -433,45 +439,42 @@ "Cloud architects generating infrastructure-as-code templates" ], "strengths": [ - "Comprehensive access to AWS cloud services and infrastructure", - "Built on highly reliable AWS infrastructure (99.99% uptime)", - "Powerful automation capabilities for cloud operations", - "Excellent audit logging through CloudTrail", - "Granular IAM permissions for access control", - "Supports major AWS services (S3, EC2, Lambda, RDS, DynamoDB)" + "Official first-party AWS offering: managed AWS MCP Server (GA 2026-05) plus the open-source awslabs/mcp suite", + "Managed server exposes 15,000+ AWS API operations through a small fixed tool set (call_aws, documentation search/read, sandboxed run_script)", + "IAM SigV4 authentication with support for read-only restriction via IAM policies or Service Control Policies", + "Excellent audit logging through CloudTrail plus CloudWatch metrics in the AWS-MCP namespace", + "Documented separation of human vs agent permissions for governance", + "No charge for the managed server itself; pay only for resources created" ], "limitations": [ - "CRITICAL SECURITY RISK: AI can create, modify, delete infrastructure", + "CRITICAL SECURITY RISK: with write-capable IAM permissions the AI can create, modify, delete infrastructure", "Cost control risk - AI can provision expensive resources", "Data in S3, RDS, and other services exposed to LLM provider", "No built-in PII detection or sensitive data filtering", - "Complex IAM setup required for secure operation", - "Potential for catastrophic infrastructure changes if misconfigured" + "Managed server requires the local MCP Proxy for AWS since MCP clients speak OAuth 2.1, not SigV4", + "Managed endpoint regions limited (US East N. Virginia, Europe Frankfurt), though API calls can target any region", + "Potential for catastrophic infrastructure changes if IAM is misconfigured" ], "metadata": { - "license": "Varies (AWS SDK Apache 2.0, MCP implementation varies)", + "license": "Apache 2.0 (awslabs/mcp suite and MCP Proxy for AWS); managed server is a hosted AWS service", "supported_platforms": [ - "All platforms with Node.js/Python" + "All platforms with Python/uvx (awslabs servers); any MCP client via MCP Proxy for AWS (managed server)" ], "programming_languages": [ - "TypeScript", - "JavaScript", - "Python" + "Python", + "TypeScript" ], "mcp_version": "1.0", - "aws_sdk_version": "v3", + "remote_endpoint": "https://aws-mcp.us-east-1.api.aws/mcp (via local MCP Proxy for AWS)", + "github_repo": "https://github.com/awslabs/mcp", "supported_services": [ - "S3", - "EC2", - "Lambda", - "RDS", - "DynamoDB", - "CloudWatch", - "IAM" + "15,000+ AWS API operations via managed server (call_aws)", + "Service-scoped awslabs servers (DynamoDB, CDK, pricing, Step Functions, S3 Tables, etc.)" ], - "authentication": "AWS Access Keys or IAM Roles", + "authentication": "AWS IAM credentials (SigV4) bridged to OAuth via MCP Proxy for AWS; IAM roles or access keys for self-hosted servers", + "ga_date": "2026-05-06", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "AWS" }, "tags": [ "aws", diff --git a/data/mcps/mcp-server-azure.json b/data/mcps/mcp-server-azure.json index 4910277..3f703ac 100644 --- a/data/mcps/mcp-server-azure.json +++ b/data/mcps/mcp-server-azure.json @@ -3,11 +3,11 @@ "type": "mcp", "name": "MCP Azure Server", "provider": "Microsoft", - "version": "1.0.0", - "last_evaluated": "2025-11-09", + "version": "2.0 (GA)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official Microsoft MCP server for Azure cloud services integration. Enables AI models to interact with Azure resources including Virtual Machines, Storage, Databases, App Services, and monitoring. Essential for AI-powered Azure infrastructure management and DevOps workflows.", - "website": "https://github.com/azure/azure-mcp-server", + "description": "Official Microsoft MCP server for Azure cloud services, generally available as Azure MCP Server 2.0 (2026-04-10) with tools spanning 44+ Azure service areas including compute, storage, databases, AI/ML, monitoring, and governance. Development now lives in the microsoft/mcp monorepo (the former Azure/azure-mcp repo is archived). Installable via npm (@azure/mcp), MCPB bundles for Claude Desktop, VS Code, NuGet, PyPI, and Docker, with a self-hosted remote HTTP mode.", + "website": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", "trust_vector": { "performance_reliability": { "overall_score": 84, @@ -24,7 +24,7 @@ } ], "methodology": "API stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_operation_success": { "score": 85, @@ -32,13 +32,13 @@ "evidence": [ { "source": "Azure MCP Server", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", + "url": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", + "date": "2026-07-09", "value": "High success rate for resource management operations" } ], "methodology": "Operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_region_performance": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Geographic performance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -74,13 +74,13 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", + "url": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", + "date": "2026-07-09", "value": "Handles Azure API errors with retry and fallback logic" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -88,18 +88,18 @@ "overall_score": 72, "criteria": { "authentication_security": { - "score": 82, + "score": 84, "confidence": "high", "evidence": [ { - "source": "Azure Authentication", - "url": "https://docs.microsoft.com/en-us/azure/active-directory/develop/", - "date": "2025-11-16", - "value": "Uses Azure AD with service principals or managed identities" + "source": "Azure MCP Server README (microsoft/mcp)", + "url": "https://github.com/microsoft/mcp/blob/main/servers/Azure.Mcp.Server/README.md", + "date": "2026-07-09", + "value": "Credentials handled through the official Azure Identity SDK: Azure CLI (az login), environment credentials, managed identity, and Microsoft Entra ID; no raw secrets in MCP client config required" } ], - "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "methodology": "Authentication mechanism review; small raise from 82 reflecting standardized Azure Identity SDK credential chain in the GA release", + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 65, @@ -113,7 +113,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rbac_enforcement": { "score": 80, @@ -127,7 +127,7 @@ } ], "methodology": "Authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "destructive_operation_risk": { "score": 60, @@ -135,13 +135,13 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", + "url": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", + "date": "2026-07-09", "value": "AI can delete resources, modify configurations, and manage infrastructure" } ], "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 85, @@ -155,7 +155,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "network_security": { "score": 75, @@ -169,7 +169,7 @@ } ], "methodology": "Network security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sensitive_data_in_resources": { "score": 65, @@ -196,13 +196,13 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", + "url": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", + "date": "2026-07-09", "value": "May expose connection strings, keys, and secrets in resource configurations" } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance_boundary_control": { "score": 72, @@ -216,7 +216,7 @@ } ], "methodology": "Compliance boundary assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_residency_considerations": { "score": 75, @@ -244,12 +244,12 @@ } ], "methodology": "Data residency review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "trust_transparency": { - "overall_score": 83, + "overall_score": 86, "criteria": { "documentation_quality": { "score": 85, @@ -257,13 +257,13 @@ "evidence": [ { "source": "Azure MCP Docs", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", + "url": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", + "date": "2026-07-09", "value": "Comprehensive documentation with Azure-specific examples" } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -277,7 +277,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 90, @@ -285,46 +285,46 @@ "evidence": [ { "source": "GitHub Repository", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", + "url": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", + "date": "2026-07-09", "value": "Open source implementation from Microsoft" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { - "score": 70, - "confidence": "medium", + "score": 80, + "confidence": "high", "evidence": [ { - "source": "MCP Server Documentation", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", - "value": "Clear but limited documentation of supported Azure services" + "source": "Azure MCP Server README (microsoft/mcp)", + "url": "https://github.com/microsoft/mcp/blob/main/servers/Azure.Mcp.Server/README.md", + "date": "2026-07-09", + "value": "GA documentation enumerates supported service areas and tools with per-service documentation in the microsoft/mcp monorepo" } ], - "methodology": "API documentation review", - "last_verified": "2025-11-09" + "methodology": "API documentation review; raised from 70 as GA docs now enumerate the tool surface", + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 80, + "overall_score": 81, "criteria": { "ease_of_setup": { - "score": 75, - "confidence": "medium", + "score": 80, + "confidence": "high", "evidence": [ { - "source": "Setup Documentation", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", - "value": "Requires Azure AD app registration and service principal setup" + "source": "Azure MCP Server README (microsoft/mcp)", + "url": "https://github.com/microsoft/mcp/blob/main/servers/Azure.Mcp.Server/README.md", + "date": "2026-07-09", + "value": "Installable via npm (@azure/mcp), drag-and-drop MCPB bundles for Claude Desktop (no runtime required), VS Code integration, NuGet, PyPI, and Docker; authenticates with existing az login session" } ], - "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "methodology": "Setup complexity assessment; raised from 75 due to MCPB bundles and az-login-based auth removing service principal setup for interactive use", + "last_verified": "2026-07-09" }, "api_performance": { "score": 82, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -352,35 +352,35 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "service_coverage": { - "score": 78, - "confidence": "medium", + "score": 84, + "confidence": "high", "evidence": [ { - "source": "Azure MCP Server", - "url": "https://github.com/azure/azure-mcp-server", - "date": "2025-11-16", - "value": "Covers major Azure services but not comprehensive" + "source": "Azure MCP Server README (microsoft/mcp)", + "url": "https://github.com/microsoft/mcp/blob/main/servers/Azure.Mcp.Server/README.md", + "date": "2026-07-09", + "value": "2.0 GA covers 44+ Azure service areas across compute, storage, databases, AI/ML, monitoring, and governance" } ], - "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "methodology": "Feature coverage assessment; raised from 78 as GA 2.0 substantially expanded service coverage", + "last_verified": "2026-07-09" }, "community_adoption": { "score": 72, "confidence": "medium", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/azure/azure-mcp-server/discussions", - "date": "2025-11-16", - "value": "Growing adoption among Azure users" + "source": "microsoft/mcp Repository", + "url": "https://github.com/microsoft/mcp", + "date": "2026-07-09", + "value": "Active first-party development in the microsoft/mcp monorepo with frequent releases; growing adoption via VS Code, Claude Desktop MCPB bundles, and Visual Studio integration" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -433,34 +433,36 @@ "Educators teaching Azure and cloud computing concepts" ], "strengths": [ - "Official Microsoft implementation with Azure expertise", - "Comprehensive Azure resource management capabilities", + "Official Microsoft implementation, generally available as 2.0 since April 2026", + "Broad Azure resource management: tools across 44+ Azure service areas", "Built on reliable Azure APIs with strong SLAs", "Full operation auditability through Azure Activity Log", - "Open source with Microsoft support", - "Supports Azure RBAC for granular permissions" + "Open source (MIT) in the microsoft/mcp monorepo with Microsoft support", + "Supports Azure RBAC for granular permissions; credentials via the official Azure Identity SDK", + "Flexible distribution: npm, MCPB bundles, VS Code/Visual Studio integration, NuGet, PyPI, Docker, self-hosted remote HTTP" ], "limitations": [ "Azure resource metadata and configurations exposed to LLM provider", - "Risk of destructive operations (resource deletion, network changes)", - "May expose connection strings, keys, and secrets", - "Complex setup requiring Azure AD configuration", - "Limited to supported Azure services (not comprehensive)", + "Risk of destructive operations (resource deletion, network changes); Microsoft warns autonomous or misconfigured clients may perform destructive actions and recommends least-privilege roles", + "May expose connection strings, keys, and secrets in resource configurations", + "Agent inherits the full RBAC permissions of the authenticated identity", + "Not every Azure service is covered despite broad 2.0 GA coverage", "Data residency considerations with LLM provider data sharing" ], "metadata": { "license": "MIT", "supported_platforms": [ - "All platforms with Node.js/Python" + "Windows, macOS, Linux (x64/ARM64) via npm, MCPB bundles, NuGet, PyPI, Docker" ], "programming_languages": [ - "TypeScript", - "Python" + "C# (.NET)" ], "mcp_version": "1.0", - "github_repo": "https://github.com/azure/azure-mcp-server", - "api_dependency": "Azure REST API, Azure SDK", - "authentication": "Azure AD, Service Principal, Managed Identity", + "github_repo": "https://github.com/microsoft/mcp/tree/main/servers/Azure.Mcp.Server", + "previous_repo": "https://github.com/Azure/azure-mcp (archived; development moved to microsoft/mcp)", + "api_dependency": "Azure REST API, Azure SDK, Azure Identity SDK", + "authentication": "Microsoft Entra ID via Azure Identity SDK (az login, environment credentials, managed identity)", + "ga_date": "2026-04-10", "first_release": "2024-11", "maintained_by": "Microsoft" }, diff --git a/data/mcps/mcp-server-brave-search.json b/data/mcps/mcp-server-brave-search.json index e303070..8a37c7f 100644 --- a/data/mcps/mcp-server-brave-search.json +++ b/data/mcps/mcp-server-brave-search.json @@ -4,9 +4,9 @@ "name": "MCP Brave Search Server", "provider": "Anthropic (Archived)", "version": "2025.4.24", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former Anthropic reference MCP server for the Brave Search API, archived 2025-05-29 to the servers-archived repository and no longer maintained (no security guarantees). Brave now ships its own official Brave Search MCP server, which is the recommended replacement. The underlying Brave Search API remains active and privacy-focused.", + "description": "ARCHIVED: Former Anthropic reference MCP server for the Brave Search API, archived 2025-05-29 to the servers-archived repository and no longer maintained (no security guarantees). Brave now ships its own official Brave Search MCP server (@brave/brave-search-mcp-server on npm, v2.0.85 as of June 2026), which is the recommended replacement. The underlying Brave Search API remains active and privacy-focused.", "website": "https://brave.com/search/api/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Search quality benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_response_time": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "API latency testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "result_freshness": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "Content freshness testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 58, @@ -86,7 +86,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -105,7 +105,7 @@ } ], "methodology": "Authentication security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_sanitization": { "score": 88, @@ -119,7 +119,7 @@ } ], "methodology": "Input sanitization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_exposure_risk": { "score": 82, @@ -133,7 +133,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "malicious_content_protection": { "score": 80, @@ -147,7 +147,7 @@ } ], "methodology": "Content safety assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -166,7 +166,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_retention": { "score": 90, @@ -180,7 +180,7 @@ } ], "methodology": "Data retention policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_sharing": { "score": 85, @@ -194,7 +194,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_anonymization": { "score": 88, @@ -208,7 +208,7 @@ } ], "methodology": "Anonymization assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -227,7 +227,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_transparency": { "score": 80, @@ -241,7 +241,7 @@ } ], "methodology": "Algorithmic transparency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_logging": { "score": 82, @@ -255,7 +255,7 @@ } ], "methodology": "Query visibility assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "implementation_clarity": { "score": 62, @@ -272,10 +272,16 @@ "url": "https://github.com/modelcontextprotocol/servers-archived", "date": "2026-06-10", "value": "Code now lives in the read-only servers-archived repository; README states no security guarantees are provided for archived servers" + }, + { + "source": "npm: @brave/brave-search-mcp-server", + "url": "https://www.npmjs.com/package/@brave/brave-search-mcp-server", + "date": "2026-07-09", + "value": "Brave's official Brave Search MCP server is actively maintained; v2.0.85 published June 2026 with web, image, video, news, local, and summarizer search plus STDIO/HTTP transports" } ], "methodology": "Code transparency review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -294,7 +300,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_reliability": { "score": 85, @@ -308,7 +314,7 @@ } ], "methodology": "Uptime monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cost_efficiency": { "score": 82, @@ -322,7 +328,7 @@ } ], "methodology": "Pricing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "result_formatting": { "score": 80, @@ -336,7 +342,7 @@ } ], "methodology": "Response format assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 78, @@ -350,7 +356,7 @@ } ], "methodology": "Feature completeness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -417,7 +423,7 @@ "Smaller index compared to Google, may miss some niche content", "API costs can accumulate with heavy usage", "Limited advanced search features compared to Google", - "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; use Brave's official Brave Search MCP server instead" + "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; use Brave's official @brave/brave-search-mcp-server (actively maintained, v2.0.85 June 2026) instead" ], "metadata": { "license": "Proprietary API, MCP server varies", @@ -436,7 +442,7 @@ "first_release": "2024-11", "maintained_by": "None (Archived 2025-05-29)", "repository": "https://github.com/modelcontextprotocol/servers-archived", - "replacement": "Brave's official Brave Search MCP server" + "replacement": "Brave's official Brave Search MCP server (@brave/brave-search-mcp-server, https://github.com/brave/brave-search-mcp-server)" }, "tags": [ "search", diff --git a/data/mcps/mcp-server-calendar.json b/data/mcps/mcp-server-calendar.json index f0351f3..71dcbf5 100644 --- a/data/mcps/mcp-server-calendar.json +++ b/data/mcps/mcp-server-calendar.json @@ -4,9 +4,9 @@ "name": "MCP Google Calendar Server", "provider": "Community", "version": "1.0.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Google Calendar integration. Enables AI models to create, read, update, and delete calendar events, manage attendees, set reminders, and query availability. Essential for AI-powered scheduling, meeting coordination, and calendar automation workflows.", + "description": "Community-maintained MCP server for Google Calendar integration. Enables AI models to create, read, update, and delete calendar events, manage attendees, set reminders, and query availability. NOTE: Google now offers an official hosted Calendar MCP server (calendarmcp.googleapis.com) via the Workspace Developer Preview Program, announced May 2026; it supersedes community Calendar MCP servers for most new deployments and is the recommended option where available.", "website": "https://github.com/modelcontextprotocol/servers", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Operation success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "availability_search_accuracy": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Availability accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sync_reliability": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Sync reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "recurring_event_handling": { "score": 83, @@ -66,7 +66,7 @@ } ], "methodology": "Recurrence pattern testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 68, @@ -113,7 +113,7 @@ } ], "methodology": "Token security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "event_modification_risk": { "score": 65, @@ -127,7 +127,7 @@ } ], "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "attendee_invitation_risk": { "score": 70, @@ -141,7 +141,7 @@ } ], "methodology": "Invitation risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "calendar_sharing_control": { "score": 78, @@ -155,7 +155,7 @@ } ], "methodology": "ACL enforcement testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 78, @@ -169,7 +169,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "attendee_privacy": { "score": 63, @@ -202,7 +202,7 @@ } ], "methodology": "Attendee privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "schedule_pattern_exposure": { "score": 70, @@ -216,7 +216,7 @@ } ], "methodology": "Pattern privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "location_privacy": { "score": 72, @@ -244,7 +244,7 @@ } ], "methodology": "Location privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -277,7 +277,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -291,7 +291,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "privacy_risk_disclosure": { "score": 72, @@ -305,7 +305,7 @@ } ], "methodology": "Privacy disclosure review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_performance": { "score": 85, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 88, @@ -352,7 +352,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage": { "score": 82, @@ -366,7 +366,7 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 75, @@ -377,10 +377,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Community-maintained with moderate activity" + }, + { + "source": "Google Workspace MCP servers documentation", + "url": "https://developers.google.com/workspace/guides/configure-mcp-servers", + "date": "2026-07-09", + "value": "Google launched an official Calendar MCP server (https://calendarmcp.googleapis.com/mcp/v1) with OAuth 2.0 in the Workspace Developer Preview Program (announced May 2026); community Calendar MCP servers are superseded for most uses" } ], "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -446,7 +452,8 @@ "Can send meeting invitations to arbitrary attendees", "Calendar patterns may reveal personal routines and habits", "Requires Google Cloud project and OAuth configuration", - "Attendee email addresses and names exposed" + "Attendee email addresses and names exposed", + "Community implementations vary in quality and maintenance; Google's official Calendar MCP server (Workspace Developer Preview, May 2026) is the recommended option going forward" ], "metadata": { "license": "MIT", @@ -462,7 +469,8 @@ "api_dependency": "Google Calendar API, Google APIs Client Library", "authentication": "OAuth 2.0 with Google", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "Community", + "official_alternative": "Google official Calendar MCP server (https://calendarmcp.googleapis.com/mcp/v1, Workspace Developer Preview Program, announced May 2026)" }, "tags": [ "calendar", diff --git a/data/mcps/mcp-server-chrome-devtools.json b/data/mcps/mcp-server-chrome-devtools.json index d730af0..9d1a5d1 100644 --- a/data/mcps/mcp-server-chrome-devtools.json +++ b/data/mcps/mcp-server-chrome-devtools.json @@ -3,8 +3,8 @@ "type": "mcp", "name": "Chrome DevTools MCP", "provider": "Google (Chrome DevTools team)", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Google's official MCP server that gives AI agents full Chrome control via the Chrome DevTools Protocol. Around 26 tools span input automation, navigation, performance tracing and insights, network inspection, console debugging, and screenshots — making it the reference server for AI-assisted web debugging and performance work.", "website": "https://github.com/ChromeDevTools/chrome-devtools-mcp", @@ -24,7 +24,7 @@ } ], "methodology": "Review of CDP connection handling for launched and attached Chrome instances", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 87, @@ -38,7 +38,7 @@ } ], "methodology": "Hands-on testing of click, fill, navigate, and wait tools against common web applications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "performance_tracing_accuracy": { "score": 92, @@ -52,7 +52,7 @@ } ], "methodology": "Validation of trace recording and insight extraction against DevTools Performance panel output", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -66,7 +66,7 @@ } ], "methodology": "Error-path testing including navigation failures, missing elements, and dropped CDP connections", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "automation_stability": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Multi-step workflow stability testing across navigation, input, and inspection tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Threat modeling of untrusted page, console, and network content entering the agent context", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "arbitrary_code_execution_risk": { "score": 48, @@ -113,7 +113,7 @@ } ], "methodology": "Capability analysis of in-page JS evaluation under adversarial agent steering", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "session_exposure_risk": { "score": 45, @@ -127,7 +127,7 @@ } ], "methodology": "Analysis of attach-to-running-Chrome mode and access to authenticated browser state", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sandboxing_isolation": { "score": 60, @@ -141,7 +141,7 @@ } ], "methodology": "Review of isolation options (fresh profile, headless, channel selection) versus attach mode", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 55, @@ -155,7 +155,7 @@ } ], "methodology": "Authorization boundary analysis of write-capable browser actions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of inspection tool outputs (network, console, screenshot, snapshot)", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 55, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy controls assessment of network and console inspection outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_data_control": { "score": 78, @@ -202,7 +202,7 @@ } ], "methodology": "Review of local execution model and data residency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing pathway analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -249,7 +249,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -263,7 +263,7 @@ } ], "methodology": "Logging and observability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "vendor_credibility": { "score": 95, @@ -272,12 +272,12 @@ { "source": "GitHub API", "url": "https://api.github.com/repos/ChromeDevTools/chrome-devtools-mcp", - "date": "2026-06-10", - "value": "Maintained by Google's Chrome DevTools team; 43,277 GitHub stars as of 2026-06-10, among the most-starred MCP servers" + "date": "2026-07-09", + "value": "Maintained by Google's Chrome DevTools team; 46,455 GitHub stars as of 2026-07-09, among the most-starred MCP servers" } ], "methodology": "Maintainer reputation and project health analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment across MCP hosts", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "performance": { "score": 86, @@ -310,7 +310,7 @@ } ], "methodology": "Latency and token-efficiency evaluation of common debugging workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 90, @@ -324,7 +324,7 @@ } ], "methodology": "Feature completeness assessment against web debugging and automation needs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 93, @@ -333,12 +333,12 @@ { "source": "GitHub API", "url": "https://api.github.com/repos/ChromeDevTools/chrome-devtools-mcp", - "date": "2026-06-10", - "value": "43,277 stars as of 2026-06-10; widely adopted by coding agents for in-browser verification and performance debugging since its 2025 launch" + "date": "2026-07-09", + "value": "46,455 stars as of 2026-07-09; widely adopted by coding agents for in-browser verification and performance debugging since its 2025 launch" } ], "methodology": "Adoption metrics and ecosystem-integration analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "maintenance_activity": { "score": 92, @@ -347,19 +347,19 @@ { "source": "GitHub repository activity", "url": "https://github.com/ChromeDevTools/chrome-devtools-mcp/releases", - "date": "2026-06-10", - "value": "Regular releases tracking Chrome versions with active issue triage by the Chrome DevTools team" + "date": "2026-07-09", + "value": "Regular releases tracking Chrome versions (chrome-devtools-mcp at v1.5.0 on npm, repo pushed 2026-07-09) with active issue triage by the Chrome DevTools team" } ], "methodology": "Commit frequency and release-cadence analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } }, "strengths": [ "First-party Chrome DevTools data: real performance traces, insights, network and console inspection", - "Maintained by Google's Chrome DevTools team with very high adoption (43.3k stars)", + "Maintained by Google's Chrome DevTools team with very high adoption (46.5k stars)", "~26 well-documented tools spanning automation, debugging, tracing, and screenshots", "Direct CDP control is fast and exposes capabilities no other browser MCP offers", "Runs fully locally over stdio with no vendor telemetry", @@ -383,8 +383,9 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/ChromeDevTools/chrome-devtools-mcp", - "github_stars": 43277, + "github_stars": 46455, "package": "chrome-devtools-mcp", + "package_version": "1.5.0", "api_dependency": "Chrome DevTools Protocol (via Puppeteer)", "authentication": "None required (local browser control)", "first_release": "2025-09", diff --git a/data/mcps/mcp-server-cloudflare.json b/data/mcps/mcp-server-cloudflare.json index 9781aaa..670726d 100644 --- a/data/mcps/mcp-server-cloudflare.json +++ b/data/mcps/mcp-server-cloudflare.json @@ -3,11 +3,11 @@ "type": "mcp", "name": "MCP Cloudflare Server", "provider": "Cloudflare", - "version": "1.0.0", - "last_evaluated": "2025-11-09", + "version": "2026.7 (managed remote servers)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official Cloudflare MCP server for managing CDN, DNS, Workers, Pages, and security services. Enables AI models to configure edge computing, manage domains, deploy serverless functions, and control security policies. Essential for AI-powered edge infrastructure and web performance optimization.", - "website": "https://github.com/cloudflare/cloudflare-mcp-server", + "description": "Official Cloudflare catalog of managed remote MCP servers, connected over OAuth. The primary Cloudflare API server at https://mcp.cloudflare.com/mcp exposes 2,500+ API endpoints (DNS, Workers, R2, Zero Trust, and more) through two Code Mode tools, search() and execute(), in roughly 1,000 tokens. Sixteen additional product-specific servers cover documentation, Workers bindings and builds, observability, Radar, browser rendering, AI Gateway, audit logs, and more.", + "website": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", "trust_vector": { "performance_reliability": { "overall_score": 86, @@ -24,7 +24,7 @@ } ], "methodology": "API stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "dns_operation_success": { "score": 90, @@ -32,13 +32,13 @@ "evidence": [ { "source": "Cloudflare MCP Server", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "High success rate for DNS, CDN, and Workers operations" } ], "methodology": "Operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "global_edge_performance": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Geographic performance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -74,46 +74,46 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "Handles API errors with retry logic and fallback" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 70, + "overall_score": 73, "criteria": { "authentication_security": { - "score": 80, + "score": 85, "confidence": "high", "evidence": [ { - "source": "Cloudflare API Authentication", - "url": "https://developers.cloudflare.com/fundamentals/api/get-started/keys/", - "date": "2025-11-16", - "value": "Uses API tokens with scoped permissions or legacy API keys" + "source": "Cloudflare Managed MCP Servers Documentation", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", + "value": "Managed remote servers use an OAuth flow where the user is redirected to Cloudflare to authorize and select permissions; API token bearer auth remains available for CI/CD" } ], - "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "methodology": "Authentication mechanism review; raised from 80 as the default connection model is now OAuth with explicit permission selection rather than static API tokens", + "last_verified": "2026-07-09" }, "token_exposure_risk": { - "score": 65, + "score": 74, "confidence": "high", "evidence": [ { - "source": "MCP Security Model", - "url": "https://modelcontextprotocol.io/docs/security", - "date": "2025-11-16", - "value": "Cloudflare API token stored locally; AI can perform actions within scope" + "source": "Cloudflare Managed MCP Servers Documentation", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", + "value": "Remote OAuth connections avoid storing long-lived API tokens in client configuration; token exposure risk remains for CI/CD bearer-token setups" } ], - "methodology": "Token security analysis", - "last_verified": "2025-11-09" + "methodology": "Token security analysis; raised from 65 because the remote OAuth default removes locally stored long-lived tokens", + "last_verified": "2026-07-09" }, "dns_manipulation_risk": { "score": 58, @@ -121,13 +121,13 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "AI can modify DNS records, potentially causing service disruption" } ], "methodology": "Operation risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "security_policy_modification": { "score": 62, @@ -135,13 +135,13 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "Can modify firewall rules, WAF settings, and security policies" } ], "methodology": "Security policy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 85, @@ -155,7 +155,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "zone_access_control": { "score": 75, @@ -169,7 +169,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "domain_information_disclosure": { "score": 68, @@ -196,13 +196,13 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "Domain names, origins, and infrastructure details may be exposed" } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "analytics_data_privacy": { "score": 75, @@ -216,7 +216,7 @@ } ], "methodology": "Data privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 72, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "secrets_in_workers": { "score": 70, @@ -238,18 +238,18 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "Workers code may contain secrets or sensitive configuration" } ], "methodology": "Secrets exposure assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "trust_transparency": { - "overall_score": 84, + "overall_score": 86, "criteria": { "documentation_quality": { "score": 87, @@ -257,13 +257,13 @@ "evidence": [ { "source": "Cloudflare MCP Docs", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "Comprehensive documentation with Cloudflare-specific examples" } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -277,7 +277,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 90, @@ -285,46 +285,46 @@ "evidence": [ { "source": "GitHub Repository", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", "value": "Open source implementation from Cloudflare" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { - "score": 72, - "confidence": "medium", + "score": 80, + "confidence": "high", "evidence": [ { - "source": "MCP Server Documentation", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", - "value": "Clear but limited documentation of supported Cloudflare services" + "source": "Cloudflare Managed MCP Servers Documentation", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", + "value": "Official docs enumerate every managed server with its endpoint URL and purpose; SSE transport is deprecated in favor of streamable-http at /mcp paths" } ], - "methodology": "API documentation review", - "last_verified": "2025-11-09" + "methodology": "API documentation review; raised from 72 as the current docs enumerate the full managed server catalog", + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 83, + "overall_score": 85, "criteria": { "ease_of_setup": { - "score": 82, + "score": 88, "confidence": "high", "evidence": [ { - "source": "Setup Documentation", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", - "value": "Requires Cloudflare API token with appropriate zone permissions" + "source": "Cloudflare Managed MCP Servers Documentation", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", + "value": "Zero-install: add the managed server URL (e.g. https://mcp.cloudflare.com/mcp) in an OAuth-capable MCP client and authorize; no local server or API token required for interactive use" } ], - "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "methodology": "Setup complexity assessment; raised from 82 as managed remote servers eliminate local installation and token provisioning", + "last_verified": "2026-07-09" }, "api_performance": { "score": 88, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 90, @@ -352,35 +352,35 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "service_coverage": { - "score": 78, - "confidence": "medium", + "score": 86, + "confidence": "high", "evidence": [ { - "source": "Cloudflare MCP Server", - "url": "https://github.com/cloudflare/cloudflare-mcp-server", - "date": "2025-11-16", - "value": "Covers DNS, Workers, Pages, CDN, and security features" + "source": "Cloudflare Managed MCP Servers Documentation", + "url": "https://developers.cloudflare.com/agents/model-context-protocol/mcp-servers-for-cloudflare/", + "date": "2026-07-09", + "value": "Primary API server exposes 2,500+ endpoints via Code Mode search()/execute(); 16 product-specific servers cover docs, bindings, builds, observability, Radar, containers, browser rendering, Logpush, AI Gateway, AI Search, audit logs, DNS analytics, DEX, CASB, and GraphQL" } ], - "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "methodology": "Feature coverage assessment; raised from 78 as coverage expanded to essentially the full Cloudflare API plus a broad server catalog", + "last_verified": "2026-07-09" }, "community_adoption": { "score": 77, "confidence": "medium", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/cloudflare/cloudflare-mcp-server/discussions", - "date": "2025-11-16", - "value": "Growing adoption among Cloudflare users and edge developers" + "source": "cloudflare/mcp-server-cloudflare Repository", + "url": "https://github.com/cloudflare/mcp-server-cloudflare", + "date": "2026-07-09", + "value": "Growing adoption among Cloudflare users and edge developers; Cloudflare also publishes templates and skills for building remote MCP servers on Workers" } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -433,34 +433,37 @@ "Educators teaching serverless and edge computing concepts" ], "strengths": [ - "Official Cloudflare implementation with edge computing expertise", - "Comprehensive edge infrastructure management (DNS, CDN, Workers, security)", - "Excellent global API performance with 300+ edge locations", + "Official Cloudflare managed remote MCP servers with OAuth and per-connection permission selection", + "Token-efficient Code Mode: 2,500+ API endpoints reachable through search()/execute() in about 1,000 tokens", + "Broad catalog of 16 product-specific servers (docs, observability, Radar, bindings, audit logs, and more)", + "Excellent global API performance on Cloudflare's edge network", "Full operation auditability through Cloudflare audit logs", - "Open source with Cloudflare support", - "High reliability with 99.95%+ uptime" + "Open source implementations with Cloudflare support; high reliability with 99.95%+ uptime" ], "limitations": [ "Infrastructure configurations and Workers code exposed to LLM provider", "Risk of DNS manipulation causing service disruption", - "Can modify security policies and firewall rules", + "Can modify security policies and firewall rules within authorized permissions", "Domain names and infrastructure details may be disclosed", "Workers code may contain secrets or sensitive configuration", - "Requires careful API token scoping to limit access" + "execute() on the primary API server can reach any authorized endpoint, so OAuth permission selection must be scoped carefully; CI/CD bearer tokens still need manual scoping" ], "metadata": { - "license": "MIT", + "license": "Apache 2.0 (cloudflare/mcp-server-cloudflare)", "supported_platforms": [ - "All platforms with Node.js/Python" + "Any OAuth-capable MCP client (remote managed servers); self-hostable on Cloudflare Workers" ], "programming_languages": [ - "TypeScript", - "Python" + "TypeScript" ], "mcp_version": "1.0", - "github_repo": "https://github.com/cloudflare/cloudflare-mcp-server", + "github_repo": "https://github.com/cloudflare/mcp-server-cloudflare", + "remote_endpoint": "https://mcp.cloudflare.com/mcp (plus 16 product-specific *.mcp.cloudflare.com/mcp endpoints)", + "transport_types": [ + "streamable-http (SSE deprecated)" + ], "api_dependency": "Cloudflare API v4", - "authentication": "Cloudflare API Token or API Key", + "authentication": "OAuth (interactive, with permission selection) or Cloudflare API token bearer auth (CI/CD)", "first_release": "2024-11", "maintained_by": "Cloudflare" }, diff --git a/data/mcps/mcp-server-context7.json b/data/mcps/mcp-server-context7.json index eaebd3d..a623122 100644 --- a/data/mcps/mcp-server-context7.json +++ b/data/mcps/mcp-server-context7.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "Context7 MCP", "provider": "Upstash", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Upstash's documentation-retrieval MCP server. Two tools (resolve-library-id, get-library-docs) inject up-to-date, version-specific library documentation into the agent's context to prevent hallucinated APIs. The most-starred MCP server repo (57.1k), but with a notable security history: the ContextCrush content-injection vulnerability (disclosed Feb 2026, patched within days).", + "description": "Upstash's documentation-retrieval MCP server. Two tools (resolve-library-id, get-library-docs) inject up-to-date, version-specific library documentation into the agent's context to prevent hallucinated APIs. The most-starred MCP server repo (58.8k), but with a notable security history: the ContextCrush content-injection vulnerability (disclosed Feb 2026, patched within days).", "website": "https://context7.com", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Spot-check of returned documentation against official library docs across popular frameworks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "retrieval_relevance": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Relevance assessment of topic-filtered retrievals across common and long-tail queries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_reliability": { "score": 86, @@ -52,7 +52,7 @@ } ], "methodology": "Availability monitoring of the hosted API endpoint over the evaluation period", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "library_coverage": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Coverage sampling across mainstream and long-tail open-source libraries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Error-path testing with unknown libraries, rate limits, and offline backend", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -105,7 +105,7 @@ } ], "methodology": "Review of the ContextCrush vulnerability, its patch, and the residual risk of doc-content prompt injection", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "supply_chain_trust": { "score": 50, @@ -119,7 +119,7 @@ } ], "methodology": "Analysis of the open library-submission pipeline as an attack surface for agent contexts", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "vulnerability_response": { "score": 75, @@ -133,7 +133,7 @@ } ], "methodology": "Assessment of disclosure-to-patch timeline and vendor cooperation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 55, @@ -147,7 +147,7 @@ } ], "methodology": "Analysis of indirect credential-theft pathways via injected instructions in a multi-tool agent", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "authentication_security": { "score": 70, @@ -161,7 +161,7 @@ } ], "methodology": "Review of API-key handling for the hosted endpoint and local server", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -180,7 +180,7 @@ } ], "methodology": "Data flow analysis of outbound query content to the hosted backend", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 68, @@ -194,7 +194,7 @@ } ], "methodology": "Assessment of direct and indirect sensitive-data pathways", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_minimization": { "score": 85, @@ -208,7 +208,7 @@ } ], "methodology": "Review of request payloads and tool surface area", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 72, @@ -222,7 +222,7 @@ } ], "methodology": "Data sharing pathway analysis across Upstash and LLM provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -241,7 +241,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -255,7 +255,7 @@ } ], "methodology": "Source availability review of client/server versus hosted backend", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "incident_disclosure": { "score": 80, @@ -269,7 +269,7 @@ } ], "methodology": "Review of vendor communication during and after the ContextCrush incident", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 85, @@ -283,7 +283,7 @@ } ], "methodology": "Logging and traceability assessment of tool calls and returned content", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -302,7 +302,7 @@ } ], "methodology": "Setup complexity assessment across remote and local installation paths", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "performance": { "score": 85, @@ -316,7 +316,7 @@ } ], "methodology": "Latency measurement of resolve and retrieval calls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 80, @@ -330,7 +330,7 @@ } ], "methodology": "Feature scope assessment relative to documentation-retrieval needs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 95, @@ -339,12 +339,12 @@ { "source": "GitHub API", "url": "https://api.github.com/repos/upstash/context7", - "date": "2026-06-10", - "value": "57,127 stars as of 2026-06-10 — the largest MCP server repository on GitHub; integrated into setup guides of most major MCP clients" + "date": "2026-07-09", + "value": "58,801 stars as of 2026-07-09 — the largest MCP server repository on GitHub; integrated into setup guides of most major MCP clients" } ], "methodology": "Adoption metrics and ecosystem-integration analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "maintenance_activity": { "score": 90, @@ -353,19 +353,19 @@ { "source": "GitHub repository activity", "url": "https://github.com/upstash/context7/commits", - "date": "2026-06-10", - "value": "Active maintenance by Upstash with regular releases, fast security patching (ContextCrush fixed in 5 days), and continuous library-index growth" + "date": "2026-07-09", + "value": "Active maintenance by Upstash with regular releases (@upstash/context7-mcp at v3.2.3, repo pushed 2026-07-09), fast security patching (ContextCrush fixed in 5 days), and continuous library-index growth" } ], "methodology": "Commit frequency, release cadence, and patch-responsiveness analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } }, "strengths": [ "Directly addresses hallucinated/outdated APIs with version-specific, current documentation", - "Largest MCP server community on GitHub (57.1k stars) with broad client integration", + "Largest MCP server community on GitHub (58.8k stars) with broad client integration", "Minimal tool surface: two read-only tools with a small, predictable data footprint", "Zero-friction setup — remote endpoint works with just a URL, no API key required", "Fast vendor response to the ContextCrush vulnerability (patched in 5 days)", @@ -389,8 +389,9 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/upstash/context7", - "github_stars": 57127, + "github_stars": 58801, "package": "@upstash/context7-mcp", + "package_version": "3.2.3", "remote_endpoint": "https://mcp.context7.com/mcp", "api_dependency": "Context7 hosted documentation API (Upstash)", "authentication": "Optional API key (higher rate limits)", diff --git a/data/mcps/mcp-server-datadog.json b/data/mcps/mcp-server-datadog.json index 7b0772b..b3cbefd 100644 --- a/data/mcps/mcp-server-datadog.json +++ b/data/mcps/mcp-server-datadog.json @@ -3,11 +3,11 @@ "type": "mcp", "name": "MCP Datadog Server", "provider": "Datadog", - "version": "1.0.0", - "last_evaluated": "2025-11-09", + "version": "hosted remote (GA 2026-03)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official Datadog MCP server for observability and monitoring integration. Enables AI models to query metrics, logs, traces, dashboards, and alerts. Includes infrastructure monitoring, APM, and security monitoring. Essential for AI-powered observability, performance optimization, and incident management workflows.", - "website": "https://github.com/DataDog/datadog-mcp-server", + "description": "Official Datadog MCP server for observability integration, generally available since March 2026 as a hosted streamable-HTTP remote (US1: https://mcp.datadoghq.com/api/unstable/mcp-server/mcp, with per-site regional endpoints) plus a local binary. Authenticates via OAuth 2.0 (recommended), access-token bearer header, or DD_API_KEY/DD_APPLICATION_KEY headers. Queries metrics, logs, traces, dashboards, and alerts; apm, code-exec, and remote-actions toolsets remain in preview.", + "website": "https://docs.datadoghq.com/mcp_server/", "trust_vector": { "performance_reliability": { "overall_score": 88, @@ -24,7 +24,7 @@ } ], "methodology": "Query success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "log_search_accuracy": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Search accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "real_time_monitoring": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Real-time monitoring testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 86, @@ -74,13 +74,13 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Handles API errors with exponential backoff and retry" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -96,10 +96,16 @@ "url": "https://docs.datadoghq.com/account_management/api-app-keys/", "date": "2025-11-16", "value": "Uses API keys and application keys with scoped permissions" + }, + { + "source": "Datadog Docs - Set Up the MCP Server", + "url": "https://docs.datadoghq.com/mcp_server/setup/", + "date": "2026-07-09", + "value": "Hosted remote supports OAuth 2.0 (recommended, no long-lived credentials), personal/service access token bearer auth, or scoped DD_API_KEY/DD_APPLICATION_KEY headers from a service account" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_key_exposure_risk": { "score": 65, @@ -113,7 +119,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "infrastructure_visibility_risk": { "score": 62, @@ -121,13 +127,13 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "AI can access detailed infrastructure topology and configurations" } ], "methodology": "Visibility risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "monitor_modification_control": { "score": 75, @@ -141,7 +147,7 @@ } ], "methodology": "Modification control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "organization_access_control": { "score": 80, @@ -155,7 +161,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 88, @@ -169,7 +175,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +194,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "log_data_privacy": { "score": 62, @@ -196,13 +202,13 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Application logs may contain PII, credentials, and sensitive data" } ], "methodology": "Log privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "infrastructure_metadata_privacy": { "score": 68, @@ -210,13 +216,13 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Infrastructure topology and host information may be exposed" } ], "methodology": "Infrastructure privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -230,7 +236,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "trace_data_privacy": { "score": 72, @@ -244,7 +250,7 @@ } ], "methodology": "Trace privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -257,13 +263,13 @@ "evidence": [ { "source": "Datadog MCP Docs", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Comprehensive documentation from official Datadog team" } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -277,7 +283,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -285,13 +291,13 @@ "evidence": [ { "source": "GitHub Repository", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Open source implementation from Datadog" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 75, @@ -299,13 +305,13 @@ "evidence": [ { "source": "MCP Server Documentation", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Clear documentation of supported Datadog operations" } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -318,13 +324,19 @@ "evidence": [ { "source": "Setup Documentation", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Simple setup requiring Datadog API and application keys" + }, + { + "source": "Datadog Docs - Set Up the MCP Server", + "url": "https://docs.datadoghq.com/mcp_server/setup/", + "date": "2026-07-09", + "value": "Hosted remote requires no install (streamable HTTP + OAuth); 15+ documented client integrations and a local binary fallback for restricted environments" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_performance": { "score": 88, @@ -338,7 +350,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 92, @@ -352,7 +364,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage": { "score": 85, @@ -360,13 +372,19 @@ "evidence": [ { "source": "Datadog MCP Server", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Covers metrics, logs, traces, dashboards, monitors, and infrastructure" + }, + { + "source": "Datadog Docs - MCP Server", + "url": "https://docs.datadoghq.com/mcp_server/", + "date": "2026-07-09", + "value": "Core toolsets GA; apm, code-exec, and remote-actions toolsets remain in preview and require signup" } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "official_support": { "score": 90, @@ -374,13 +392,19 @@ "evidence": [ { "source": "Datadog Team", - "url": "https://github.com/DataDog/datadog-mcp-server", + "url": "https://github.com/datadog-labs/mcp-server", "date": "2025-11-16", "value": "Officially maintained by Datadog with active support" + }, + { + "source": "Datadog Press Release - MCP Server GA", + "url": "https://www.datadoghq.com/about/latest-news/press-releases/datadog-launches-mcp-server/", + "date": "2026-03-09", + "value": "Datadog MCP Server announced generally available, providing AI agents secure real-time access to unified observability data" } ], "methodology": "Maintainer support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -436,6 +460,7 @@ "strengths": [ "Comprehensive observability platform (metrics, logs, traces, infrastructure)", "Official Datadog implementation with active support", + "Hosted remote GA since 2026-03 with OAuth 2.0 and per-site regional endpoints", "Highly reliable with 99.9%+ uptime on global infrastructure", "Excellent for AI-powered performance optimization and incident management", "Real-time alerting and monitoring with low latency", @@ -447,28 +472,38 @@ "Infrastructure topology and host information may be revealed", "Distributed traces may contain request parameters and user identifiers", "Subject to Datadog API rate limits", - "Requires careful log scrubbing to avoid sensitive data exposure" + "Requires careful log scrubbing to avoid sensitive data exposure", + "Some toolsets (apm, code-exec, remote-actions) still in preview and require signup" ], "metadata": { - "license": "MIT", + "license": "Hosted service (see datadog-labs/mcp-server repo for source terms)", "supported_platforms": [ - "All platforms with Node.js/Python" + "Hosted remote (per-site endpoints, e.g. mcp.datadoghq.com)", + "Local binary (macOS, Linux, Windows)" ], "programming_languages": [ - "TypeScript", - "Python" + "N/A (hosted service; local binary available)" ], "mcp_version": "1.0", - "github_repo": "https://github.com/DataDog/datadog-mcp-server", + "github_repo": "https://github.com/datadog-labs/mcp-server", "api_dependency": "Datadog REST API", - "authentication": "Datadog API Key and Application Key", + "authentication": "OAuth 2.0 (recommended), access token bearer header, or DD_API_KEY + DD_APPLICATION_KEY headers", + "remote_endpoint": "https://mcp.datadoghq.com/api/unstable/mcp-server/mcp (US1; per-site regional endpoints, e.g. mcp.datadoghq.eu)", + "remote_ga_date": "2026-03-09", "first_release": "2024-11", - "maintained_by": "Datadog" + "maintained_by": "Datadog", + "status": "Active - official Datadog MCP Server GA since 2026-03", + "transport_types": [ + "streamable-http (hosted remote)", + "stdio (local binary)" + ] }, "tags": [ "monitoring", "observability", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "remote-server" ] } diff --git a/data/mcps/mcp-server-docker.json b/data/mcps/mcp-server-docker.json index 8f149bc..06b3220 100644 --- a/data/mcps/mcp-server-docker.json +++ b/data/mcps/mcp-server-docker.json @@ -4,10 +4,10 @@ "name": "MCP Docker Server", "provider": "Community", "version": "1.0.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Docker container management. Enables AI models to interact with Docker Engine for container lifecycle management, image operations, volume management, and network configuration. Essential for AI-powered containerized application deployment and DevOps automation.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "Community-maintained MCP server for Docker container management; the leading implementation is ckreiling/mcp-server-docker (Python, GPL-3.0, ~728 stars), which manages containers via the Docker SDK and deliberately blocks sensitive options like --privileged and --cap-add. Covers containers, images, volumes, and networks. Note: Docker Inc. has NOT shipped an official engine-management MCP server; its official MCP Catalog, Toolkit, and Gateway distribute containerized MCP servers.", + "website": "https://github.com/ckreiling/mcp-server-docker", "trust_vector": { "performance_reliability": { "overall_score": 85, @@ -24,7 +24,7 @@ } ], "methodology": "API stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "container_operation_success": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_monitoring_accuracy": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Monitoring accuracy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "image_pull_reliability": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Image operation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Access control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "container_escape_risk": { "score": 60, @@ -110,10 +110,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "AI can create privileged containers with host access" + }, + { + "source": "ckreiling/mcp-server-docker README", + "url": "https://github.com/ckreiling/mcp-server-docker", + "date": "2026-07-09", + "value": "Leading community implementation deliberately does not support --privileged or --cap-add/--cap-drop; README still warns Docker is not a secure sandbox and the server can impact the host through the Docker socket" } ], "methodology": "Privilege escalation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "image_security_scanning": { "score": 70, @@ -127,7 +133,7 @@ } ], "methodology": "Security scanning assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "network_isolation_control": { "score": 68, @@ -141,7 +147,7 @@ } ], "methodology": "Network security testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "secret_exposure_risk": { "score": 62, @@ -155,7 +161,7 @@ } ], "methodology": "Secrets management assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 75, @@ -169,7 +175,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +194,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "log_privacy": { "score": 62, @@ -202,7 +208,7 @@ } ], "methodology": "Log privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "environment_variable_protection": { "score": 68, @@ -216,7 +222,7 @@ } ], "methodology": "Environment security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -230,7 +236,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "volume_data_access": { "score": 72, @@ -244,7 +250,7 @@ } ], "methodology": "Storage privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +269,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -277,7 +283,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -288,10 +294,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Open source community implementation" + }, + { + "source": "ckreiling/mcp-server-docker", + "url": "https://github.com/ckreiling/mcp-server-docker", + "date": "2026-07-09", + "value": "Leading community implementation is open source (Python, GPL-3.0) with ~728 stars; supports local socket and remote Docker daemons over SSH" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "security_warning_disclosure": { "score": 72, @@ -302,10 +314,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Some security warnings but could be more prominent" + }, + { + "source": "ckreiling/mcp-server-docker README", + "url": "https://github.com/ckreiling/mcp-server-docker", + "date": "2026-07-09", + "value": "README carries explicit, prominent warnings: 'DO NOT CONFIGURE CONTAINERS WITH SENSITIVE DATA' and cautions to review LLM-created containers because Docker is not a secure sandbox" } ], "methodology": "Security disclosure review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +342,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_performance": { "score": 83, @@ -338,7 +356,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -352,7 +370,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 80, @@ -366,7 +384,7 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 75, @@ -380,7 +398,7 @@ } ], "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -442,29 +460,31 @@ ], "limitations": [ "Requires Docker socket access equivalent to root privileges", - "AI can create privileged containers with host system access", + "AI can start containers with host access via the socket (leading implementation blocks --privileged/--cap-add, but the socket itself remains root-equivalent)", "Container logs and environment variables exposed to LLM provider", "No built-in image vulnerability scanning", "Can access and expose Docker secrets and sensitive configurations", - "Community-maintained with variable support quality" + "Community-maintained with variable support quality", + "No official Docker Inc. engine-management MCP server exists; Docker's official MCP products (MCP Catalog, Toolkit, Gateway) address server distribution/orchestration instead" ], "metadata": { - "license": "MIT", + "license": "GPL-3.0 (ckreiling/mcp-server-docker)", "supported_platforms": [ "Linux", "macOS", "Windows (with Docker Desktop)" ], "programming_languages": [ - "TypeScript", "Python" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", - "api_dependency": "Docker Engine API", - "authentication": "Docker socket access", - "first_release": "2024-11", - "maintained_by": "Community" + "github_repo": "https://github.com/ckreiling/mcp-server-docker", + "github_stars": 728, + "api_dependency": "Docker Engine API (Docker SDK for Python)", + "authentication": "Docker socket access (local) or Docker daemon over SSH (remote)", + "first_release": "2024-12", + "maintained_by": "Community", + "official_vendor_alternative": "Docker MCP Catalog and Toolkit / MCP Gateway (server distribution and orchestration, not engine management): https://docs.docker.com/ai/mcp-catalog-and-toolkit/" }, "tags": [ "containers", diff --git a/data/mcps/mcp-server-elasticsearch.json b/data/mcps/mcp-server-elasticsearch.json index 646f344..182fc01 100644 --- a/data/mcps/mcp-server-elasticsearch.json +++ b/data/mcps/mcp-server-elasticsearch.json @@ -2,12 +2,12 @@ "id": "mcp-server-elasticsearch", "type": "mcp", "name": "MCP Elasticsearch Server", - "provider": "Community", - "version": "1.0.0", - "last_evaluated": "2025-11-09", + "provider": "Elastic", + "version": "0.4.6 (deprecated; successor: Agent Builder MCP)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Elasticsearch search and analytics operations. Enables AI models to perform full-text search, aggregations, indexing, and data analytics on large-scale datasets. Essential for AI-powered search optimization, log analysis, and business intelligence workflows.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "Official Elastic MCP server (elastic/mcp-server-elasticsearch) exposing read-oriented tools - list_indices, get_mappings, search, esql, get_shards - for full-text search, ES|QL, and analytics. IMPORTANT: the standalone server is deprecated (v0.4.6, 2025-10-24, critical security updates only); Elastic directs users to the Agent Builder MCP endpoint ({KIBANA_URL}/api/agent_builder/mcp), GA in Elastic 9.2+ and Elasticsearch Serverless, with API key authentication.", + "website": "https://github.com/elastic/mcp-server-elasticsearch", "trust_vector": { "performance_reliability": { "overall_score": 84, @@ -24,7 +24,7 @@ } ], "methodology": "Search relevance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "aggregation_reliability": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Aggregation accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "large_index_performance": { "score": 78, @@ -52,7 +52,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cluster_stability": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Cluster stability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -74,18 +74,18 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", "value": "Handles Elasticsearch errors with retry and timeout management" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 69, + "overall_score": 75, "criteria": { "authentication_security": { "score": 78, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 62, @@ -113,49 +113,49 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_injection_risk": { - "score": 65, + "score": 68, "confidence": "high", "evidence": [ { - "source": "Security Analysis", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "AI can construct arbitrary queries and scripts" + "source": "elastic/mcp-server-elasticsearch Repository", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", + "value": "AI can construct arbitrary search and ES|QL queries; blast radius is limited to reads since no write tools exist, but expensive queries and broad data reads remain possible" } ], "methodology": "Injection vulnerability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "index_modification_risk": { - "score": 60, + "score": 82, "confidence": "high", "evidence": [ { - "source": "Elasticsearch API", - "url": "https://www.elastic.co/guide/en/elasticsearch/reference/current/docs.html", - "date": "2025-11-16", - "value": "AI can index, update, and delete documents within permissions" + "source": "elastic/mcp-server-elasticsearch Repository", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", + "value": "The official server ships only read-oriented tools (list_indices, get_mappings, search, esql, get_shards); it exposes no document indexing, update, or write tools" } ], - "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "methodology": "Operation authorization testing; raised from 60 because the official Elastic server's tool surface is read-only, unlike earlier community servers with write tools", + "last_verified": "2026-07-09" }, "index_deletion_risk": { - "score": 68, - "confidence": "medium", + "score": 85, + "confidence": "high", "evidence": [ { - "source": "Implementation Review", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Can delete indices if permissions allow" + "source": "elastic/mcp-server-elasticsearch Repository", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", + "value": "No index deletion or index management tools are exposed by the official server" } ], - "methodology": "Destructive operation testing", - "last_verified": "2025-11-09" + "methodology": "Destructive operation testing; raised from 68 as the official tool surface contains no deletion capability", + "last_verified": "2026-07-09" }, "audit_logging": { "score": 80, @@ -169,7 +169,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_in_logs": { "score": 60, @@ -196,13 +196,13 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", "value": "Log data and application logs may contain PII" } ], "methodology": "PII exposure assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "field_level_security": { "score": 72, @@ -216,7 +216,7 @@ } ], "methodology": "Field security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "index_pattern_exposure": { "score": 70, @@ -238,13 +238,13 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", "value": "Index names and mappings may reveal data structure" } ], "methodology": "Index privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -257,13 +257,13 @@ "evidence": [ { "source": "Elasticsearch MCP Docs", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", "value": "Good documentation but community-maintained with evolving coverage" } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_visibility": { "score": 82, @@ -277,7 +277,7 @@ } ], "methodology": "Query logging assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -285,13 +285,13 @@ "evidence": [ { "source": "GitHub Repository", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Open source community implementation" + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", + "value": "Official Elastic implementation, open source under Apache 2.0 (Rust, ~686 stars)" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 70, @@ -299,18 +299,18 @@ "evidence": [ { "source": "MCP Server Documentation", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", "value": "Clear but incomplete documentation of supported Elasticsearch APIs" } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 81, + "overall_score": 76, "criteria": { "ease_of_setup": { "score": 80, @@ -318,13 +318,13 @@ "evidence": [ { "source": "Setup Documentation", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", "value": "Requires Elasticsearch connection URL and credentials" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_performance": { "score": 82, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 83, @@ -352,35 +352,35 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage": { - "score": 80, + "score": 72, "confidence": "high", "evidence": [ { - "source": "Elasticsearch MCP Server", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Covers search, aggregations, indexing, and index management" + "source": "elastic/mcp-server-elasticsearch Repository", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", + "value": "Deliberately narrow read-only surface: search, ES|QL, index listing, mappings, and shard info; no indexing or index management tools. Broader/custom tool coverage now comes from the Agent Builder MCP endpoint" } ], - "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "methodology": "Feature coverage assessment; lowered from 80 as the official server's surface is narrower than the previously described community coverage", + "last_verified": "2026-07-09" }, "community_support": { - "score": 75, - "confidence": "medium", + "score": 62, + "confidence": "high", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/modelcontextprotocol/servers/discussions", - "date": "2025-11-16", - "value": "Community-maintained with moderate activity" + "source": "elastic/mcp-server-elasticsearch Repository", + "url": "https://github.com/elastic/mcp-server-elasticsearch", + "date": "2026-07-09", + "value": "Repository carries a deprecation notice: only critical security updates going forward; users are directed to the Agent Builder MCP endpoint in Elastic 9.2+ / Serverless" } ], - "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "methodology": "Community support assessment; lowered from 75 due to the standalone server's deprecation", + "last_verified": "2026-07-09" } } } @@ -434,36 +434,42 @@ "Educators teaching search technology and data analytics" ], "strengths": [ - "Highly accurate full-text search with relevance scoring", - "Powerful aggregation framework for analytics and insights", - "Built on mature Elasticsearch client libraries", + "Highly accurate full-text search with relevance scoring, plus ES|QL query support", + "Official Elastic implementation, open source under Apache 2.0", + "Read-only tool surface limits blast radius (no write or index management tools)", "Excellent for log analysis and business intelligence", - "Open source community implementation", - "Comprehensive audit logging in X-Pack Security" + "Supports stdio and streamable-HTTP transports", + "Clear migration path to the GA Agent Builder MCP endpoint with custom tools" ], "limitations": [ + "DEPRECATED: standalone server receives only critical security updates; Elastic directs users to the Agent Builder MCP endpoint (Elastic 9.2+/Serverless)", "Search results and indexed documents exposed to LLM provider", "Log data may contain PII and sensitive information", - "AI can modify and delete indices within permission scope", - "Elasticsearch credentials accessible to AI", - "Performance issues with very large indices without optimization", - "Field-level security requires X-Pack and careful configuration" + "Elasticsearch credentials/API keys stored in local client configuration", + "Arbitrary search and ES|QL queries can be expensive on large indices", + "Field-level security requires platform security features and careful configuration" ], "metadata": { - "license": "MIT", + "license": "Apache 2.0", "supported_platforms": [ - "All platforms with Node.js/Python" + "All platforms (Rust binary, Docker); Agent Builder MCP via Kibana endpoint" ], "programming_languages": [ - "TypeScript", - "Python" + "Rust" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", - "api_dependency": "Elasticsearch client libraries", - "authentication": "API keys, Basic auth, X-Pack Security", - "first_release": "2024-11", - "maintained_by": "Community" + "github_repo": "https://github.com/elastic/mcp-server-elasticsearch", + "status": "Deprecated (critical security updates only) as of the v0.4.x line", + "successor": "Elastic Agent Builder MCP endpoint: {KIBANA_URL}/api/agent_builder/mcp (GA in Elastic 9.2+ and Elasticsearch Serverless)", + "latest_release": "v0.4.6 (2025-10-24)", + "transport_types": [ + "stdio", + "streamable-http" + ], + "api_dependency": "Elasticsearch REST API", + "authentication": "Elasticsearch API keys or username/password; Agent Builder endpoint uses Kibana API keys", + "first_release": "2025", + "maintained_by": "Elastic" }, "tags": [ "search", diff --git a/data/mcps/mcp-server-everything.json b/data/mcps/mcp-server-everything.json index bd914d0..e1e0dcf 100644 --- a/data/mcps/mcp-server-everything.json +++ b/data/mcps/mcp-server-everything.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Everything Server", "provider": "Anthropic", - "version": "2025.9.25", - "last_evaluated": "2026-06-10", + "version": "2026.7.4", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official MCP reference server demonstrating all protocol features (tools, resources, prompts, sampling) for testing and development. One of the seven reference servers still actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest spec 2025-11-25. Not intended for production use but essential for MCP protocol understanding and testing.", + "description": "Official MCP reference server demonstrating all protocol features (tools, resources, prompts, sampling) for testing and development. One of seven reference servers still maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest stable spec 2025-11-25 (2026-07-28 revision at release-candidate stage). Not for production use but essential for MCP protocol testing.", "website": "https://modelcontextprotocol.io/docs/servers/everything", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "protocol_compliance": { "score": 92, @@ -38,7 +38,7 @@ } ], "methodology": "Protocol compliance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "test_reliability": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Testing reliability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "example_accuracy": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Example validation", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_readiness": { "score": 50, @@ -80,7 +80,7 @@ } ], "methodology": "Production readiness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Data security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "capability_exposure": { "score": 60, @@ -113,7 +113,7 @@ } ], "methodology": "Capability exposure testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "production_deployment_risk": { "score": 55, @@ -127,7 +127,7 @@ } ], "methodology": "Deployment risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "input_validation": { "score": 75, @@ -141,7 +141,7 @@ } ], "methodology": "Input validation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_limits": { "score": 65, @@ -155,7 +155,7 @@ } ], "methodology": "Resource limit testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data usage analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "test_data_exposure": { "score": 70, @@ -188,7 +188,7 @@ } ], "methodology": "Data exposure analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "development_privacy": { "score": 78, @@ -202,7 +202,7 @@ } ], "methodology": "Development privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "code_example_quality": { "score": 98, @@ -249,7 +249,7 @@ } ], "methodology": "Code quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -263,7 +263,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "protocol_feature_documentation": { "score": 92, @@ -277,7 +277,7 @@ } ], "methodology": "Feature documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "testing_utility": { "score": 95, @@ -310,7 +310,7 @@ } ], "methodology": "Testing utility assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "development_value": { "score": 92, @@ -324,7 +324,7 @@ } ], "methodology": "Developer value assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "protocol_education": { "score": 90, @@ -338,7 +338,7 @@ } ], "methodology": "Educational value assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 55, @@ -355,10 +355,16 @@ "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", "date": "2025-12-09", "value": "everything is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" + }, + { + "source": "npm - @modelcontextprotocol/server-everything", + "url": "https://www.npmjs.com/package/@modelcontextprotocol/server-everything", + "date": "2026-07-09", + "value": "Latest release 2026.7.4 (published 2026-07-04, built on MCP SDK 1.29.0); still listed among the seven maintained reference servers in modelcontextprotocol/servers README (repo ~88,300 stars, not archived)" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -425,7 +431,7 @@ "May expose all protocol capabilities without restrictions", "Not optimized for performance or scale", "Requires careful consideration before any production use", - "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" + "STATUS 2026-07-09: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); latest npm release 2026.7.4 (2026-07-04); governance under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -438,7 +444,7 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/modelcontextprotocol/servers", - "github_stars": 58700, + "github_stars": 88259, "api_dependency": "None (MCP protocol only)", "authentication": "None required", "first_release": "2024-11", diff --git a/data/mcps/mcp-server-fetch.json b/data/mcps/mcp-server-fetch.json index 31d60d6..d387b03 100644 --- a/data/mcps/mcp-server-fetch.json +++ b/data/mcps/mcp-server-fetch.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Fetch Server", "provider": "Anthropic", - "version": "2025.4.6", - "last_evaluated": "2026-06-10", + "version": "2026.6.4", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official MCP reference server for fetching web content and converting HTML to markdown. Enables AI models to retrieve and process web pages, documentation, and online resources. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", + "description": "Official MCP reference server for fetching web content and converting HTML to markdown. Enables AI models to retrieve and process web pages, documentation, and online resources. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest stable spec 2025-11-25; 2026-07-28 spec revision at release-candidate stage as of 2026-07).", "website": "https://modelcontextprotocol.io/docs/servers/fetch", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Fetch operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "html_to_markdown_accuracy": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Conversion quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "timeout_handling": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Timeout behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Rate limiting behavior analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 83, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "SSRF vulnerability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "url_validation": { "score": 75, @@ -113,7 +113,7 @@ } ], "methodology": "Input validation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ssl_certificate_validation": { "score": 85, @@ -127,7 +127,7 @@ } ], "methodology": "SSL/TLS security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "malicious_content_protection": { "score": 68, @@ -141,7 +141,7 @@ } ], "methodology": "Content security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sensitive_data_exposure": { "score": 70, @@ -155,7 +155,7 @@ } ], "methodology": "Data flow security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "user_agent_disclosure": { "score": 72, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy disclosure review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cookie_handling": { "score": 70, @@ -202,7 +202,7 @@ } ], "methodology": "Cookie management assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "paywalled_content_respect": { "score": 65, @@ -230,7 +230,7 @@ } ], "methodology": "Content access policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -263,7 +263,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -277,7 +277,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "security_disclosure": { "score": 85, @@ -291,7 +291,7 @@ } ], "methodology": "Security documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -310,7 +310,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "fetch_performance": { "score": 83, @@ -324,7 +324,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -338,7 +338,7 @@ } ], "methodology": "Uptime and stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversion_quality": { "score": 87, @@ -352,7 +352,7 @@ } ], "methodology": "Conversion accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 85, @@ -369,10 +369,16 @@ "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", "date": "2025-12-09", "value": "fetch is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" + }, + { + "source": "PyPI - mcp-server-fetch", + "url": "https://pypi.org/project/mcp-server-fetch/", + "date": "2026-07-09", + "value": "Latest release 2026.6.4 (published 2026-06-04); still listed among the seven maintained reference servers in modelcontextprotocol/servers README (repo ~88,300 stars, not archived)" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -392,7 +398,7 @@ "No malicious content scanning or sanitization", "Performance dependent on target website responsiveness", "Basic rate limiting may cause issues with aggressive scraping", - "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" + "STATUS 2026-07-09: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); latest PyPI release 2026.6.4 (2026-06-04); no CVEs against this reference implementation (note: CVE-2025-65513 SSRF affects the unrelated third-party npm package 'mcp-fetch-server', not this Python mcp-server-fetch); governance under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -405,7 +411,7 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/modelcontextprotocol/servers", - "github_stars": 58700, + "github_stars": 88259, "api_dependency": "HTTP/HTTPS, Turndown", "authentication": "None required", "first_release": "2024-11", @@ -415,7 +421,7 @@ "stdio" ], "installation_methods": [ - "npm" + "pip" ] }, "use_case_ratings": { diff --git a/data/mcps/mcp-server-figma.json b/data/mcps/mcp-server-figma.json index 9d1e298..697bd98 100644 --- a/data/mcps/mcp-server-figma.json +++ b/data/mcps/mcp-server-figma.json @@ -4,9 +4,9 @@ "name": "Figma MCP Server", "provider": "Figma", "version": "2025.6-beta", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Figma's official Dev Mode MCP server connecting AI coding tools to design files. Provides design-context extraction (code from frames, variables, components), screenshots, metadata, Code Connect mapping, FigJam reading, and design generation onto the canvas. Available as a hosted remote server (OAuth) or via the Figma desktop app.", + "description": "Figma's official Dev Mode MCP server connecting AI coding tools to design files. Provides design-context extraction (code, variables, components), screenshots, metadata, Code Connect mapping, FigJam reading, and design generation onto the canvas (canvas writes expanded March 2026). Available as a hosted remote server (OAuth, all plans) or via the Figma desktop app (Dev/Full seat on paid plans). Still in beta as of July 2026; free during the beta with per-plan tool-call limits.", "website": "https://developers.figma.com/docs/figma-mcp-server/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Assessment of design-to-code output fidelity against source frames, variables, and component structure", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_reliability": { "score": 82, @@ -38,7 +38,7 @@ } ], "methodology": "Analysis of endpoint stability and Figma platform uptime during beta period", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "large_file_handling": { "score": 72, @@ -52,7 +52,7 @@ } ], "methodology": "Testing context extraction on large, deeply nested design files and component libraries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 78, @@ -66,21 +66,21 @@ } ], "methodology": "Error handling testing across invalid selections, permissions, and disconnected desktop sessions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 76, "confidence": "medium", "evidence": [ { - "source": "Figma Developers Platform", - "url": "https://developers.figma.com/docs/figma-mcp-server/", - "date": "2026-06-10", - "value": "Hosted server applies Figma platform rate limits per authenticated user; limits are not fully published during beta and usage-based pricing is planned" + "source": "Figma Developers Platform - Rate Limits & Access", + "url": "https://developers.figma.com/docs/figma-mcp-server/rate-limits-access/", + "date": "2026-07-09", + "value": "Limits are now published per plan/seat: 6 tool calls/month on Starter or View/Collab seats, 200 calls/day on Professional/Organization Full or Dev seats, 600 calls/day on Enterprise; per-minute limits also apply to read tools, while write-to-canvas tools are exempt" } ], "methodology": "Rate limiting behavior observation under sustained tool-call load", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Review of OAuth flow, scope grants, and token lifecycle for the hosted endpoint", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 80, @@ -113,7 +113,7 @@ } ], "methodology": "Token storage and exposure-surface analysis for remote and local transports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scope_limitation": { "score": 78, @@ -127,7 +127,7 @@ } ], "methodology": "Permission boundary testing across files the authenticated user can view or edit", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_risk": { "score": 58, @@ -141,7 +141,7 @@ } ], "methodology": "Threat modeling of untrusted design-file content flowing into agent context via design-context and FigJam tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 68, @@ -155,7 +155,7 @@ } ], "methodology": "Authorization boundary testing of write-capable tools against editable files", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis from Figma files through MCP tool results to LLM providers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 66, @@ -188,7 +188,7 @@ } ], "methodology": "Assessment of filtering and redaction controls on extracted design content", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "organization_data_control": { "score": 78, @@ -202,7 +202,7 @@ } ], "methodology": "Review of organizational access controls applicable to MCP-connected accounts", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 72, @@ -216,7 +216,7 @@ } ], "methodology": "Analysis of downstream data sharing once content leaves the Figma boundary", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 78, @@ -249,7 +249,7 @@ } ], "methodology": "Logging and traceability assessment across client and Figma file history", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 35, @@ -263,7 +263,7 @@ } ], "methodology": "Source availability and independent verifiability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 82, @@ -277,7 +277,7 @@ } ], "methodology": "Comparison of documented tool surface against observed server capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment across supported MCP clients", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 78, @@ -310,7 +310,7 @@ } ], "methodology": "Latency observation across tool types and frame sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 76, @@ -324,7 +324,7 @@ } ], "methodology": "Stability assessment over the beta period including breaking-change frequency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 86, @@ -338,7 +338,7 @@ } ], "methodology": "Feature completeness assessment against design-to-code workflow needs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 84, @@ -352,7 +352,7 @@ } ], "methodology": "Adoption analysis across MCP client ecosystems and developer tooling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -370,13 +370,14 @@ "Shared design files are third-party-authored input and a prompt injection vector", "Design content, including any sensitive text in mockups, is sent to the LLM provider", "Large frames can exceed client context limits", - "Beta product with evolving tools and planned usage-based pricing not yet finalized", + "Still in beta as of July 2026 with evolving tools; planned usage-based pricing not yet finalized", + "Tool-call quotas are restrictive on lower tiers (6 calls/month on Starter or View/Collab seats; 200/day on Pro/Org, 600/day on Enterprise)", "Write-capable tools can modify editable files without server-side confirmation" ], "metadata": { "license": "Proprietary (closed source)", "maintained_by": "Figma", - "status": "Beta (launched June 2025)", + "status": "Beta (launched June 2025; still beta as of July 2026, canvas write capabilities expanded March 2026)", "remote_endpoint": "https://mcp.figma.com/mcp", "local_endpoint": "http://127.0.0.1:3845/mcp (Figma desktop app)", "authentication": "OAuth (remote, recommended); desktop app session (local)", @@ -388,7 +389,7 @@ "Remote MCP endpoint", "Figma desktop app toggle" ], - "pricing": "Free during beta; usage-based pricing planned", + "pricing": "Free during beta (verified 2026-07-09); usage-based pricing planned. Remote server available on all plans/seats; desktop server requires a Dev or Full seat on a paid plan. Tool-call limits: 6/month (Starter, View/Collab seats), 200/day (Pro/Org Full or Dev seats), 600/day (Enterprise)", "first_release": "2025-06", "mcp_version": "1.0" }, diff --git a/data/mcps/mcp-server-filesystem.json b/data/mcps/mcp-server-filesystem.json index 092474c..03b3242 100644 --- a/data/mcps/mcp-server-filesystem.json +++ b/data/mcps/mcp-server-filesystem.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Filesystem Server", "provider": "Anthropic", - "version": "2025.7.1", - "last_evaluated": "2026-06-10", + "version": "2026.7.4", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official MCP reference server providing AI models with controlled access to the local filesystem (file reading, writing, and directory operations). One of the seven reference servers still actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest spec 2025-11-25. Critical for file-based workflows but requires careful security configuration.", + "description": "Official MCP reference server providing AI models with controlled access to the local filesystem (file reading, writing, and directory operations). One of seven reference servers still maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest stable spec 2025-11-25 (2026-07-28 revision at release-candidate stage). Critical for file-based workflows but requires careful security configuration.", "website": "https://modelcontextprotocol.io/docs/servers/filesystem", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "File operation testing across multiple platforms", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "response_time": { "score": 95, @@ -38,7 +38,7 @@ } ], "methodology": "Latency benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Error scenario testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cross_platform_compatibility": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Multi-platform testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "large_file_handling": { "score": 87, @@ -80,7 +80,7 @@ } ], "methodology": "Large file performance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "path_traversal_protection": { "score": 75, @@ -110,10 +110,16 @@ "url": "https://github.com/modelcontextprotocol/servers/tree/main/src/filesystem", "date": "2025-11-16", "value": "Built-in path normalization prevents basic traversal attacks" + }, + { + "source": "GitHub Advisory Database / Trend Micro disclosure", + "url": "https://github.com/advisories?query=modelcontextprotocol+server-filesystem", + "date": "2026-07-09", + "value": "CVE-2025-53110 (directory containment bypass, CVSS 7.3) and CVE-2025-53109 (symlink bypass, CVSS 8.4) were disclosed by Trend Micro in 2025 and fixed in release 2025.7.1; no newer advisories against this server found as of 2026-07-09 (current release 2026.7.4)" } ], "methodology": "Path traversal attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "permission_enforcement": { "score": 65, @@ -127,7 +133,7 @@ } ], "methodology": "Permission boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 70, @@ -141,7 +147,7 @@ } ], "methodology": "Logging capabilities review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_exfiltration_risk": { "score": 60, @@ -155,7 +161,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +180,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 65, @@ -188,7 +194,7 @@ } ], "methodology": "Privacy features assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_processing": { "score": 85, @@ -202,7 +208,7 @@ } ], "methodology": "Data processing architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 70, @@ -216,7 +222,7 @@ } ], "methodology": "Compliance framework review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -235,7 +241,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -249,7 +255,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 95, @@ -263,7 +269,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "security_guidance": { "score": 78, @@ -277,7 +283,7 @@ } ], "methodology": "Security documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -296,7 +302,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_efficiency": { "score": 92, @@ -310,7 +316,7 @@ } ], "methodology": "Resource utilization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "stability": { "score": 88, @@ -324,7 +330,7 @@ } ], "methodology": "Issue tracking analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "maintenance_requirements": { "score": 85, @@ -341,10 +347,16 @@ "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", "date": "2025-12-09", "value": "filesystem is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" + }, + { + "source": "npm - @modelcontextprotocol/server-filesystem", + "url": "https://www.npmjs.com/package/@modelcontextprotocol/server-filesystem", + "date": "2026-07-09", + "value": "Latest release 2026.7.4 (published 2026-07-04); still listed among the seven maintained reference servers in modelcontextprotocol/servers README (repo ~88,300 stars, not archived)" } ], "methodology": "Maintenance overhead assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_support": { "score": 82, @@ -358,7 +370,7 @@ } ], "methodology": "Community activity analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -374,11 +386,11 @@ "limitations": [ "Significant security risk if misconfigured - AI can access all allowed files", "No built-in PII or sensitive data detection/redaction", - "File contents sent to external LLM provider APIs", + "File contents sent to external LLM provider APIs, with potential for accidental data exfiltration", "Limited granular permission controls beyond directory allowlists", - "Potential for accidental data exfiltration to LLM providers", "Audit logging requires custom implementation for compliance needs", - "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" + "SECURITY: CVE-2025-53110 (containment bypass) and CVE-2025-53109 (symlink bypass) fixed in 2025.7.1; versions older than 2025.7.1 remain vulnerable - upgrade to current release", + "STATUS 2026-07-09: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); latest npm release 2026.7.4 (2026-07-04); governance under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -393,7 +405,7 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/modelcontextprotocol/servers", - "github_stars": 58700, + "github_stars": 88259, "first_release": "2024-11", "maintained_by": "MCP project (Agentic AI Foundation / Linux Foundation since 2025-12-09)", "status": "Official - Active", diff --git a/data/mcps/mcp-server-firecrawl.json b/data/mcps/mcp-server-firecrawl.json index 7412e7c..293ab0c 100644 --- a/data/mcps/mcp-server-firecrawl.json +++ b/data/mcps/mcp-server-firecrawl.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "Firecrawl MCP Server", "provider": "Firecrawl", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official Firecrawl MCP server giving AI models web scraping, crawling, site mapping, web search, and structured extraction capabilities. Available as an npm package (firecrawl-mcp) over stdio or as a hosted remote server authenticated with a firecrawl.dev API key.", + "description": "Official Firecrawl MCP server (firecrawl-mcp v3.x) providing web scraping, crawling, site mapping, web search, structured extraction, document parsing, a research agent, interactive browser sessions, and page-change monitors. Runs as an npm package over stdio or as the hosted server at https://mcp.firecrawl.dev/{API_KEY}/v2/mcp; a rate-limited keyless free tier (https://mcp.firecrawl.dev/v2/mcp) covers scrape, search, and interact, and hosted transport accepts OAuth bearer tokens.", "website": "https://github.com/firecrawl/firecrawl-mcp-server", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "API stability and service maturity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scrape_success_rate": { "score": 84, @@ -38,7 +38,7 @@ } ], "methodology": "Scrape success testing across diverse site types", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "crawl_completeness": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Crawl coverage assessment against known site structures", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 83, @@ -66,7 +66,7 @@ } ], "methodology": "Rate limiting behavior review from source and docs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,26 +80,26 @@ } ], "methodology": "Error handling and recovery testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 68, + "overall_score": 67, "criteria": { "authentication_security": { "score": 80, "confidence": "high", "evidence": [ { - "source": "Firecrawl MCP Setup Documentation", - "url": "https://docs.firecrawl.dev/mcp-server", - "date": "2026-06-10", - "value": "Authenticates with a firecrawl.dev API key via environment variable (stdio) or the hosted remote endpoint" + "source": "Firecrawl MCP Server Repository", + "url": "https://github.com/firecrawl/firecrawl-mcp-server", + "date": "2026-07-09", + "value": "Authenticates with a firecrawl.dev API key via environment variable (stdio) or embedded in the hosted endpoint path; hosted HTTP transport also accepts OAuth access tokens (fco_ prefix) via Authorization Bearer header; a keyless rate-limited free tier exposes scrape, search, and interact without credentials" } ], "methodology": "Authentication mechanism review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "prompt_injection_exposure": { "score": 55, @@ -113,7 +113,7 @@ } ], "methodology": "Threat modeling of untrusted web content returned to the model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ssrf_protection": { "score": 58, @@ -127,7 +127,7 @@ } ], "methodology": "SSRF risk analysis of URL-driven tool inputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_handling": { "score": 75, @@ -141,21 +141,21 @@ } ], "methodology": "API key storage and exposure analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { - "score": 70, + "score": 65, "confidence": "medium", "evidence": [ { "source": "Firecrawl MCP Server Repository", "url": "https://github.com/firecrawl/firecrawl-mcp-server", - "date": "2026-06-10", - "value": "Tools are read-oriented (scrape, crawl, search, extract) so no destructive write actions exist, but the agent can crawl arbitrary sites and consume account credits" + "date": "2026-07-09", + "value": "Core tools remain read-oriented, but v3 adds firecrawl_interact browser automation (click, type, navigate on arbitrary sites, enabling form submissions), an autonomous research agent, and monitor tools that create recurring jobs with webhooks; the agent can also crawl arbitrary sites and consume account credits" } ], - "methodology": "Capability and blast radius assessment of exposed tools", - "last_verified": "2026-06-10" + "methodology": "Capability and blast radius assessment of exposed tools; score lowered from 70 (2026-06-10) because interact browser automation and autonomous agent/monitor tools extend the surface beyond read-only retrieval", + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of scrape and crawl pipelines", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 65, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy controls assessment of returned content", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -202,7 +202,7 @@ } ], "methodology": "Data sharing and policy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_posture": { "score": 73, @@ -216,7 +216,7 @@ } ], "methodology": "Compliance documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -249,7 +249,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -258,12 +258,12 @@ { "source": "Firecrawl MCP Server Repository", "url": "https://github.com/firecrawl/firecrawl-mcp-server", - "date": "2026-06-10", - "value": "MIT-licensed open source server with 6,538 GitHub stars; backend scraping service is partially proprietary" + "date": "2026-07-09", + "value": "MIT-licensed open source server with 6,895 GitHub stars; backend scraping service is partially proprietary" } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 86, @@ -272,12 +272,12 @@ { "source": "Firecrawl MCP Server Repository", "url": "https://github.com/firecrawl/firecrawl-mcp-server", - "date": "2026-06-10", - "value": "Tool set clearly enumerated: scrape, batch scrape, crawl, map, web search, and structured extract" + "date": "2026-07-09", + "value": "Tool set clearly enumerated in v3: scrape, map, search, crawl (+status), parse, extract, agent (+status), interact browser sessions (+stop), research (papers/repos), monitor (create/list), and feedback tools" } ], "methodology": "Tool surface documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -291,12 +291,12 @@ { "source": "firecrawl-mcp npm package", "url": "https://www.npmjs.com/package/firecrawl-mcp", - "date": "2026-06-10", - "value": "Single npx command with one API key environment variable, or zero-install hosted remote endpoint" + "date": "2026-07-09", + "value": "Single npx command with one API key environment variable, zero-install hosted remote endpoint (mcp.firecrawl.dev), or fully keyless rate-limited free tier for scrape/search/interact" } ], "methodology": "Setup complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 82, @@ -310,7 +310,7 @@ } ], "methodology": "Latency characterization across tool types", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 84, @@ -324,7 +324,7 @@ } ], "methodology": "Uptime and incident history analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 88, @@ -333,12 +333,12 @@ { "source": "Firecrawl MCP Server Repository", "url": "https://github.com/firecrawl/firecrawl-mcp-server", - "date": "2026-06-10", - "value": "Covers the full web data lifecycle: discovery (map, search), retrieval (scrape, batch, crawl), and structured extraction" + "date": "2026-07-09", + "value": "Covers the full web data lifecycle: discovery (map, search), retrieval (scrape, crawl), structured extraction, document parsing, autonomous research agent, interactive browser sessions, recurring page monitors, and academic paper/repo research tools" } ], "methodology": "Feature completeness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 86, @@ -347,23 +347,23 @@ { "source": "GitHub Repository Metrics", "url": "https://github.com/firecrawl/firecrawl-mcp-server", - "date": "2026-06-10", - "value": "6,538 GitHub stars and listing in major MCP client directories indicate strong adoption" + "date": "2026-07-09", + "value": "6,895 GitHub stars, active releases (v3.22.3 published 2026-07-08), and listing in major MCP client directories indicate strong adoption" } ], "methodology": "Community activity and adoption analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } }, "strengths": [ - "Full web data toolkit: scrape, batch scrape, crawl, map, search, and structured extract", + "Full web data toolkit: scrape, crawl, map, search, structured extract, document parsing, research agent, browser sessions, and page monitors", "Handles JavaScript-heavy sites and returns clean LLM-ready markdown", "Automatic retries with exponential backoff and credit usage monitoring", - "Simple setup via npx package or hosted remote server", - "MIT-licensed open source server with strong community adoption (6,538 stars)", - "Read-only tool surface with no destructive write actions" + "Simple setup via npx package, hosted remote server, or keyless free tier", + "MIT-licensed open source server with strong community adoption (6,895 stars)", + "Hosted transport supports OAuth bearer tokens in addition to API keys" ], "limitations": [ "All fetched web content is untrusted and can carry indirect prompt injection payloads", @@ -371,7 +371,8 @@ "Leaked API key allows arbitrary credit consumption on the account", "No built-in PII or secret filtering on scraped content", "Scraped data transits Firecrawl's cloud before reaching the model", - "Heavily bot-protected sites can still fail or return partial content" + "Heavily bot-protected sites can still fail or return partial content", + "STATUS 2026-07-09: v3 interact tool adds browser automation (click, type, navigate), so the tool surface is no longer purely read-only — injected instructions could drive form submissions on third-party sites; agent and monitor tools run asynchronously and can consume credits over time" ], "metadata": { "license": "MIT", @@ -383,14 +384,16 @@ "TypeScript" ], "github_repo": "https://github.com/firecrawl/firecrawl-mcp-server", - "github_stars": 6538, + "github_stars": 6895, "package_name": "firecrawl-mcp", - "api_dependency": "Firecrawl API (firecrawl.dev)", - "authentication": "Firecrawl API key", + "package_version": "3.22.3", + "api_dependency": "Firecrawl API v2 (firecrawl.dev)", + "authentication": "Firecrawl API key (env var or URL path) or OAuth bearer token (hosted); keyless free tier for scrape/search/interact", + "remote_endpoint": "https://mcp.firecrawl.dev/{FIRECRAWL_API_KEY}/v2/mcp (keyless: https://mcp.firecrawl.dev/v2/mcp)", "maintained_by": "Firecrawl", "transport_types": [ "stdio", - "remote (hosted)" + "streamable-http (hosted)" ], "installation_methods": [ "npm", diff --git a/data/mcps/mcp-server-git.json b/data/mcps/mcp-server-git.json index 1a2a29e..4310842 100644 --- a/data/mcps/mcp-server-git.json +++ b/data/mcps/mcp-server-git.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Git Server", "provider": "Anthropic", - "version": "2025.9.25", - "last_evaluated": "2026-06-10", + "version": "2026.6.16", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official MCP reference server for Git repository operations. Enables AI models to interact with local Git repositories, perform commits, branch management, and version control operations. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", + "description": "Official MCP reference server for local Git operations (commits, branches, version control). One of the seven reference servers still maintained after the 2025-05-29 archival; governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09 (stable spec 2025-11-25; 2026-07-28 revision at RC stage). CVE-2025-68143/68144/68145 (path traversal, argument injection; published 2026-01-20) are fixed in 2025.9.25 and 2025.12.18; current release 2026.6.16 includes all fixes.", "website": "https://modelcontextprotocol.io/docs/servers/git", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "repository_parsing_accuracy": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Parsing accuracy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "large_repository_handling": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "merge_conflict_handling": { "score": 85, @@ -66,7 +66,7 @@ } ], "methodology": "Conflict handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 89, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -96,10 +96,16 @@ "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "Respects local file system permissions but AI can access any readable repo" + }, + { + "source": "The Hacker News - Three Flaws in Anthropic MCP Git Server", + "url": "https://thehackernews.com/2026/01/three-flaws-in-anthropic-mcp-git-server.html", + "date": "2026-01-20", + "value": "CVE-2025-68143 (path traversal in git_init, CVSS 8.8, fixed in 2025.9.25) and CVE-2025-68145 (path traversal bypassing the --repository restriction, CVSS 7.1, fixed in 2025.12.18); both patched in current release 2026.6.16" } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "destructive_operation_risk": { "score": 62, @@ -110,10 +116,16 @@ "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "AI can perform force pushes, branch deletions, and history rewrites" + }, + { + "source": "The Hacker News - Three Flaws in Anthropic MCP Git Server", + "url": "https://thehackernews.com/2026/01/three-flaws-in-anthropic-mcp-git-server.html", + "date": "2026-01-20", + "value": "CVE-2025-68144: argument injection via git_diff/git_checkout (CVSS 8.1) enabling arbitrary file overwrite/code execution via prompt injection; fixed in 2025.12.18, patched in current release 2026.6.16. Score unchanged: flaws are patched upstream, but they illustrate the destructive-operation risk surface" } ], "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 68, @@ -127,7 +139,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "commit_signing_support": { "score": 78, @@ -141,7 +153,7 @@ } ], "methodology": "Signing capability review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "branch_protection_respect": { "score": 70, @@ -155,7 +167,7 @@ } ], "methodology": "Protection mechanism testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +186,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sensitive_data_detection": { "score": 60, @@ -188,7 +200,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "commit_history_privacy": { "score": 68, @@ -202,7 +214,7 @@ } ], "methodology": "History privacy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -216,7 +228,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "gitignore_respect": { "score": 72, @@ -230,7 +242,7 @@ } ], "methodology": "File filtering assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +261,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 92, @@ -263,7 +275,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -277,7 +289,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "command_coverage_clarity": { "score": 82, @@ -291,7 +303,7 @@ } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -310,7 +322,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_performance": { "score": 85, @@ -324,7 +336,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 90, @@ -338,7 +350,7 @@ } ], "methodology": "Stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 87, @@ -352,7 +364,7 @@ } ], "methodology": "Feature completeness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 85, @@ -369,10 +381,16 @@ "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", "date": "2025-12-09", "value": "git is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" + }, + { + "source": "PyPI - mcp-server-git", + "url": "https://pypi.org/project/mcp-server-git/", + "date": "2026-07-09", + "value": "Latest release 2026.6.16 (published 2026-06-17); still listed among the seven maintained reference servers in modelcontextprotocol/servers README (repo ~88,300 stars, not archived)" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -387,12 +405,12 @@ ], "limitations": [ "Repository code and history exposed to LLM provider APIs", - "Risk of destructive operations (force push, branch deletion, history rewrite)", + "Risk of destructive operations (force push, branch deletion, history rewrite); no safeguards against accidental commits or pushes", "No built-in secret detection or sensitive data filtering", "Can access Git credentials stored on local system", "Performance may degrade with very large repositories", - "No safeguards against accidental commits or pushes", - "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" + "SECURITY: three CVEs published 2026-01-20 (CVE-2025-68143 path traversal in git_init; CVE-2025-68144 argument injection in git_diff/git_checkout; CVE-2025-68145 path traversal bypassing --repository); fixed in 2025.9.25 and 2025.12.18 - versions older than 2025.12.18 remain vulnerable, upgrade to current release", + "STATUS 2026-07-09: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); latest PyPI release 2026.6.16 (2026-06-17) includes all CVE fixes; governance under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -405,7 +423,7 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/modelcontextprotocol/servers", - "github_stars": 58700, + "github_stars": 88259, "api_dependency": "Git CLI", "authentication": "Uses local Git credentials", "first_release": "2024-11", @@ -415,7 +433,7 @@ "stdio" ], "installation_methods": [ - "npm" + "pip" ] }, "use_case_ratings": { diff --git a/data/mcps/mcp-server-github.json b/data/mcps/mcp-server-github.json index 1d05d55..c3b00d2 100644 --- a/data/mcps/mcp-server-github.json +++ b/data/mcps/mcp-server-github.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP GitHub Server", "provider": "GitHub (formerly Anthropic)", - "version": "2025.4.6", - "last_evaluated": "2026-06-10", + "version": "1.5.0", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "GitHub's OFFICIAL MCP server, successor to the archived Anthropic reference server. Open source (Go, MIT, 30,558 stars), distributed as a binary/Docker image or via the hosted remote at https://api.githubcopilot.com/mcp/ (GA 2025-09-04) with OAuth 2.1+PKCE. Exposes 50+ tools in configurable toolsets with read-only mode. Known prompt-injection exfiltration risk (Invariant Labs, May 2025) requires least-privilege tokens and one-repo sessions.", + "description": "GitHub's OFFICIAL MCP server, successor to the archived Anthropic reference server. Open source (Go, MIT); binary/Docker or hosted remote at https://api.githubcopilot.com/mcp/ (GA 2025-09-04, OAuth 2.1+PKCE). Since v1.5.0 (2026-06) the local stdio server has built-in OAuth (no PAT needed); releases track the 2026-01-26 MCP spec. 50+ tools in configurable toolsets with read-only mode. Prompt-injection exfiltration risk (Invariant Labs, May 2025) requires least-privilege tokens.", "website": "https://github.com/github/github-mcp-server", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "API stability and uptime analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Operation success testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 85, @@ -66,7 +66,7 @@ } ], "methodology": "Search result quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 84, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -102,10 +102,16 @@ "url": "https://github.blog/changelog/2025-09-04-remote-github-mcp-server-is-now-generally-available/", "date": "2025-09-04", "value": "Hosted remote server (https://api.githubcopilot.com/mcp/) generally available with OAuth 2.1 + PKCE authorization" + }, + { + "source": "GitHub MCP Server releases (v1.5.0)", + "url": "https://github.com/github/github-mcp-server/releases", + "date": "2026-06-27", + "value": "v1.5.0 adds built-in OAuth to the local stdio server, so a PAT is no longer required; OAuth is the recommended auth path with the token kept in memory only" } ], "methodology": "Authentication mechanism review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 70, @@ -119,7 +125,7 @@ } ], "methodology": "Token security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scope_limitation": { "score": 82, @@ -139,7 +145,7 @@ } ], "methodology": "Permission scope testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "action_auditability": { "score": 82, @@ -153,7 +159,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 60, @@ -173,7 +179,7 @@ } ], "methodology": "Authorization boundary testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -192,7 +198,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 62, @@ -212,7 +218,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "organization_data_control": { "score": 78, @@ -226,7 +232,7 @@ } ], "methodology": "Access control review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 75, @@ -240,7 +246,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -259,7 +265,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -273,7 +279,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -282,12 +288,12 @@ { "source": "GitHub MCP Server (official)", "url": "https://github.com/github/github-mcp-server", - "date": "2026-06-10", - "value": "Fully open source Go implementation with MIT license; 30,558 GitHub stars" + "date": "2026-07-09", + "value": "Fully open source Go implementation with MIT license; ~31,300 GitHub stars" } ], "methodology": "Source code review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 85, @@ -301,7 +307,7 @@ } ], "methodology": "API documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -326,7 +332,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 82, @@ -340,7 +346,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 87, @@ -354,7 +360,7 @@ } ], "methodology": "Uptime analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 88, @@ -363,12 +369,12 @@ { "source": "GitHub MCP Server (official)", "url": "https://github.com/github/github-mcp-server", - "date": "2026-06-10", - "value": "50+ tools in configurable toolsets covering repos, issues, PRs, actions, code security, and search; read-only mode supported" + "date": "2026-07-09", + "value": "50+ tools in configurable toolsets covering repos, issues, PRs, actions, code security, and search; read-only mode supported; recent releases add MCP Apps, code quality findings, and alignment with the 2026-01-26 MCP spec" } ], "methodology": "Feature completeness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 88, @@ -377,12 +383,12 @@ { "source": "GitHub MCP Server (official)", "url": "https://github.com/github/github-mcp-server", - "date": "2026-06-10", - "value": "30,558 GitHub stars; official GitHub maintenance with hosted remote generally available since 2025-09-04" + "date": "2026-07-09", + "value": "~31,300 GitHub stars; official GitHub maintenance with weekly release cadence (v1.5.0, 2026-06-27) and hosted remote generally available since 2025-09-04" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -392,8 +398,8 @@ "Built on reliable GitHub infrastructure with high uptime", "Excellent for development workflows and code collaboration", "Full operation auditability through GitHub's audit logs", - "Official GitHub-maintained open source server (Go, MIT, 30,558 stars)", - "Hosted remote option (OAuth 2.1+PKCE) generally available since 2025-09-04", + "Official GitHub-maintained open source server (Go, MIT, ~31,300 stars)", + "Hosted remote (OAuth 2.1+PKCE) GA since 2025-09-04; local stdio server has built-in OAuth since v1.5.0 (2026-06), no PAT required", "Configurable toolsets and read-only mode limit the action surface" ], "limitations": [ @@ -417,10 +423,11 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/github/github-mcp-server", - "github_stars": 30558, + "github_stars": 31300, + "latest_release": "v1.5.0 (2026-06-27)", "deprecated_repo": "https://github.com/modelcontextprotocol/servers-archived", "api_dependency": "GitHub REST API v3", - "authentication": "GitHub PAT (local) or OAuth 2.1 + PKCE (hosted remote)", + "authentication": "OAuth (local stdio, built-in since v1.5.0) or GitHub PAT (local); OAuth 2.1 + PKCE (hosted remote)", "remote_endpoint": "https://api.githubcopilot.com/mcp/", "remote_ga_date": "2025-09-04", "first_release": "2024-11", diff --git a/data/mcps/mcp-server-gitlab.json b/data/mcps/mcp-server-gitlab.json index 7b2776f..15ba0c7 100644 --- a/data/mcps/mcp-server-gitlab.json +++ b/data/mcps/mcp-server-gitlab.json @@ -4,10 +4,10 @@ "name": "MCP GitLab Server", "provider": "Anthropic (Archived)", "version": "2025.3.2", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former Anthropic reference MCP server for GitLab integration (merge requests, CI/CD pipelines, issues), archived 2025-05-29 to the servers-archived repository and no longer maintained; no security guarantees are provided for archived servers. The GitLab MCP ecosystem has since moved to other actively maintained implementations, which should be preferred.", - "website": "https://gitlab.com/gitlab-org/gitlab-mcp-server", + "description": "ARCHIVED: Former Anthropic reference MCP server for GitLab (merge requests, CI/CD pipelines, issues), archived 2025-05-29 to servers-archived and no longer maintained; no security guarantees for archived servers. GitLab now ships an OFFICIAL built-in MCP server (experiment in GitLab 18.3, beta since 18.6) at https:///api/v4/mcp with OAuth 2.0 Dynamic Client Registration; it requires GitLab Duo and a Premium/Ultimate tier and should be preferred.", + "website": "https://github.com/modelcontextprotocol/servers-archived", "trust_vector": { "performance_reliability": { "overall_score": 80, @@ -24,7 +24,7 @@ } ], "methodology": "API stability and uptime analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Operation success testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "ci_cd_integration": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "CI/CD integration testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 60, @@ -86,7 +86,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -105,7 +105,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 72, @@ -119,7 +119,7 @@ } ], "methodology": "Token security analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "scope_limitation": { "score": 82, @@ -133,7 +133,7 @@ } ], "methodology": "Permission scope testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "action_auditability": { "score": 85, @@ -147,7 +147,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "self_hosted_option": { "score": 90, @@ -161,7 +161,7 @@ } ], "methodology": "Deployment options review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -180,7 +180,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 70, @@ -194,7 +194,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_residency": { "score": 85, @@ -208,7 +208,7 @@ } ], "methodology": "Data residency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 75, @@ -222,7 +222,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -241,7 +241,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 85, @@ -255,7 +255,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, @@ -269,7 +269,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 82, @@ -283,7 +283,7 @@ } ], "methodology": "API documentation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -302,7 +302,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "api_performance": { "score": 82, @@ -316,7 +316,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -330,7 +330,7 @@ } ], "methodology": "Uptime analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 60, @@ -347,10 +347,16 @@ "url": "https://github.com/modelcontextprotocol/servers-archived", "date": "2026-06-10", "value": "Archived 2025-05-29; feature set frozen, no updates for GitLab API or MCP specification changes" + }, + { + "source": "GitLab Docs - GitLab MCP server", + "url": "https://docs.gitlab.com/user/gitlab_duo/model_context_protocol/mcp_server/", + "date": "2026-07-09", + "value": "GitLab's official built-in MCP server (beta since GitLab 18.6) at https:///api/v4/mcp with OAuth 2.0 Dynamic Client Registration supersedes this archived server; requires GitLab Duo and Premium/Ultimate" } ], "methodology": "Feature completeness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "enterprise_features": { "score": 88, @@ -364,7 +370,7 @@ } ], "methodology": "Enterprise capabilities assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } @@ -383,19 +389,22 @@ "Some features require GitLab Premium/Ultimate", "Subject to GitLab API rate limits", "No built-in secret detection in code", - "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; GitLab ecosystem has moved to other implementations" + "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees", + "Superseded by GitLab's official built-in MCP server (beta, GitLab 18.6+, /api/v4/mcp with OAuth 2.0 DCR), which requires GitLab Duo and Premium/Ultimate tier" ], "metadata": { "license": "MIT", "supported_platforms": ["All platforms with Node.js/Python"], "programming_languages": ["TypeScript", "Python"], "mcp_version": "1.0", - "github_repo": "https://gitlab.com/gitlab-org/gitlab-mcp-server", + "github_repo": "https://github.com/modelcontextprotocol/servers-archived", "api_dependency": "GitLab REST API v4 / GraphQL", "authentication": "GitLab Personal Access Token or OAuth", "first_release": "2025-02", "maintained_by": "None (Archived 2025-05-29)", "repository": "https://github.com/modelcontextprotocol/servers-archived", + "official_successor": "GitLab built-in MCP server (beta, GitLab 18.6+): https://docs.gitlab.com/user/gitlab_duo/model_context_protocol/mcp_server/", + "successor_endpoint": "https://gitlab.com/api/v4/mcp (GitLab.com) or https:///api/v4/mcp (self-managed)", "transport_types": ["stdio"], "installation_methods": ["npm", "pip"] }, diff --git a/data/mcps/mcp-server-gmail.json b/data/mcps/mcp-server-gmail.json index 40cccbe..f82a7a1 100644 --- a/data/mcps/mcp-server-gmail.json +++ b/data/mcps/mcp-server-gmail.json @@ -4,9 +4,9 @@ "name": "MCP Gmail Server", "provider": "Community", "version": "1.0.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Gmail email operations. Enables AI models to read, send, search, label, and manage emails through the Gmail API. Includes support for attachments, threading, and advanced search. Essential for AI-powered email automation and communication workflows.", + "description": "Community-maintained MCP server for Gmail email operations. Enables AI models to read, send, search, label, and manage emails through the Gmail API. Includes support for attachments, threading, and advanced search. NOTE: Google now offers an official hosted Gmail MCP server (gmailmcp.googleapis.com) via the Workspace Developer Preview Program, announced May 2026; it supersedes community Gmail MCP servers for most new deployments and is the recommended option where available.", "website": "https://github.com/modelcontextprotocol/servers", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Search accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "send_reliability": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Send success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 78, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "attachment_handling": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Attachment processing testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 65, @@ -113,7 +113,7 @@ } ], "methodology": "Token security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "email_send_risk": { "score": 60, @@ -127,7 +127,7 @@ } ], "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "email_deletion_risk": { "score": 68, @@ -141,7 +141,7 @@ } ], "methodology": "Destructive operation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "scope_limitation": { "score": 75, @@ -155,7 +155,7 @@ } ], "methodology": "Permission scope testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 78, @@ -169,7 +169,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "contact_information_exposure": { "score": 62, @@ -202,7 +202,7 @@ } ], "methodology": "PII exposure assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "attachment_privacy": { "score": 60, @@ -216,7 +216,7 @@ } ], "methodology": "Attachment privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "conversation_history_privacy": { "score": 70, @@ -244,7 +244,7 @@ } ], "methodology": "History privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -291,7 +291,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "privacy_risk_disclosure": { "score": 70, @@ -305,7 +305,7 @@ } ], "methodology": "Privacy disclosure review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_performance": { "score": 82, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -352,7 +352,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage": { "score": 80, @@ -366,7 +366,7 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 75, @@ -377,10 +377,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Community-maintained with moderate activity" + }, + { + "source": "Google Workspace MCP servers documentation", + "url": "https://developers.google.com/workspace/guides/configure-mcp-servers", + "date": "2026-07-09", + "value": "Google launched an official Gmail MCP server (https://gmailmcp.googleapis.com/mcp/v1) with OAuth 2.0 in the Workspace Developer Preview Program (announced May 2026); community Gmail MCP servers are superseded for most uses" } ], "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -399,7 +405,8 @@ "Email addresses and contact information exposed", "Can delete or trash emails if permissions granted", "Subject to Gmail API rate limits (250 quota units/user/second)", - "Complex OAuth setup requiring Google Cloud project configuration" + "Complex OAuth setup requiring Google Cloud project configuration", + "Community implementations vary in quality and maintenance; Google's official Gmail MCP server (Workspace Developer Preview, May 2026) is the recommended option going forward" ], "metadata": { "license": "MIT", @@ -415,7 +422,8 @@ "api_dependency": "Gmail API, Google APIs Client Library", "authentication": "OAuth 2.0 with Google", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "Community", + "official_alternative": "Google official Gmail MCP server (https://gmailmcp.googleapis.com/mcp/v1, Workspace Developer Preview Program, announced May 2026)" }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-google-drive.json b/data/mcps/mcp-server-google-drive.json index 50eff47..0c61a7a 100644 --- a/data/mcps/mcp-server-google-drive.json +++ b/data/mcps/mcp-server-google-drive.json @@ -4,9 +4,9 @@ "name": "MCP Google Drive Server", "provider": "Anthropic (Archived)", "version": "1.0.0", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former Anthropic reference MCP server (gdrive) for accessing and searching Google Drive files, archived 2025-05-29 to the servers-archived repository and no longer maintained; the archive README provides no security guarantees. The underlying Google Drive API remains reliable, but this server receives no fixes and is not recommended for new deployments.", + "description": "ARCHIVED: Former Anthropic reference MCP server (gdrive) for accessing and searching Google Drive files, archived 2025-05-29 to servers-archived and no longer maintained; no security guarantees. Google now offers an official hosted Google Drive MCP server (drivemcp.googleapis.com) via the Workspace Developer Preview Program, announced May 2026, which is the recommended replacement. The archived server receives no fixes and is not recommended for new deployments.", "website": "https://developers.google.com/drive", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "File operation success rate testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Search quality testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "document_parsing": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Document conversion testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "large_file_handling": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Large file performance testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication security review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "permission_scope_control": { "score": 78, @@ -113,7 +113,7 @@ } ], "methodology": "Permission scope testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "file_access_risk": { "score": 70, @@ -127,7 +127,7 @@ } ], "methodology": "Access boundary testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sharing_control": { "score": 72, @@ -141,7 +141,7 @@ } ], "methodology": "Sharing permission testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 82, @@ -155,7 +155,7 @@ } ], "methodology": "Audit capabilities review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data exposure analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 65, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "workspace_data_control": { "score": 78, @@ -202,7 +202,7 @@ } ], "methodology": "Enterprise control assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "third_party_sharing": { "score": 72, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 80, @@ -230,7 +230,7 @@ } ], "methodology": "Compliance framework review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 85, @@ -263,7 +263,7 @@ } ], "methodology": "Operation traceability assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "mcp_implementation_clarity": { "score": 58, @@ -274,10 +274,16 @@ "url": "https://github.com/modelcontextprotocol/servers-archived", "date": "2026-06-10", "value": "Reference gdrive server archived 2025-05-29; repository read-only, README states no security guarantees are provided for archived servers" + }, + { + "source": "Google Workspace MCP servers documentation", + "url": "https://developers.google.com/workspace/guides/configure-mcp-servers", + "date": "2026-07-09", + "value": "Google now provides an official Drive MCP server (https://drivemcp.googleapis.com/mcp/v1) with OAuth 2.0, available through the Workspace Developer Preview Program; rollout announced May 2026" } ], "methodology": "Implementation documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "permission_transparency": { "score": 80, @@ -291,7 +297,7 @@ } ], "methodology": "Permission clarity assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -310,7 +316,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_reliability": { "score": 90, @@ -324,7 +330,7 @@ } ], "methodology": "Uptime analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "performance": { "score": 82, @@ -338,7 +344,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 55, @@ -358,7 +364,7 @@ } ], "methodology": "Feature completeness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost": { "score": 80, @@ -372,7 +378,7 @@ } ], "methodology": "Cost analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } } @@ -391,7 +397,7 @@ "Complex OAuth setup process requiring Google Cloud project", "AI can access and modify all files within granted permissions", "Subject to Google API rate limits and quotas", - "ARCHIVED 2025-05-29: reference gdrive server unmaintained with no security guarantees; prefer actively maintained Google Drive MCP alternatives" + "ARCHIVED 2025-05-29: reference gdrive server unmaintained with no security guarantees; prefer Google's official Drive MCP server (Workspace Developer Preview, May 2026) or other actively maintained alternatives" ], "metadata": { "license": "Varies (API proprietary, MCP implementation varies)", @@ -410,7 +416,8 @@ "rate_limits": "1000 queries per 100 seconds (default)", "first_release": "2024-11", "maintained_by": "None (Archived 2025-05-29)", - "repository": "https://github.com/modelcontextprotocol/servers-archived" + "repository": "https://github.com/modelcontextprotocol/servers-archived", + "replacement": "Google official Drive MCP server (https://drivemcp.googleapis.com/mcp/v1, Workspace Developer Preview Program)" }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-hugging-face.json b/data/mcps/mcp-server-hugging-face.json index 106c4a7..9ba4f59 100644 --- a/data/mcps/mcp-server-hugging-face.json +++ b/data/mcps/mcp-server-hugging-face.json @@ -3,8 +3,8 @@ "type": "mcp", "name": "Hugging Face MCP Server", "provider": "Hugging Face", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "0.3.28", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Hugging Face's official MCP server connecting AI assistants to the Hub. Ships 7 built-in tools (search for models, datasets, Spaces, and papers, plus documentation search) and can dynamically attach community Gradio Spaces as additional tools. Hosted at huggingface.co/mcp with per-user configuration, or runnable locally; open source under MIT.", "website": "https://huggingface.co/mcp", @@ -24,7 +24,7 @@ } ], "methodology": "Endpoint stability analysis of the hosted server and underlying Hub APIs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "Relevance assessment of Hub search results for representative ML queries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Operation success testing across built-in tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "dynamic_tool_reliability": { "score": 66, @@ -66,7 +66,7 @@ } ], "methodology": "Reliability testing of dynamically attached Gradio Space tools across popular Spaces", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing including dynamic tool set changes mid-session", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication mechanism review for hosted and local deployment modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 76, @@ -113,7 +113,7 @@ } ], "methodology": "Token storage and exposure-surface analysis across deployment modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scope_limitation": { "score": 74, @@ -127,7 +127,7 @@ } ], "methodology": "Permission scope testing of built-in tools and attached Space tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_tool_supply_chain": { "score": 50, @@ -141,7 +141,7 @@ } ], "methodology": "Supply-chain threat modeling of community Space attachment: untrusted code, mutable tool definitions, and unvetted outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 74, @@ -155,7 +155,7 @@ } ], "methodology": "Authorization boundary analysis of built-in versus attached tool capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of queries and results across the hosted server", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 70, @@ -188,7 +188,7 @@ } ], "methodology": "Assessment of filtering controls on data submitted to attached tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "organization_data_control": { "score": 72, @@ -202,7 +202,7 @@ } ], "methodology": "Access control review of Hub permissions as applied through the MCP server", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 60, @@ -216,7 +216,7 @@ } ], "methodology": "Analysis of data sharing with community Space operators and the LLM provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 78, @@ -249,7 +249,7 @@ } ], "methodology": "Logging and configuration-visibility assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 94, @@ -259,11 +259,11 @@ "source": "hf-mcp-server Repository", "url": "https://github.com/huggingface/hf-mcp-server", "date": "2026-06-10", - "value": "Server is fully open source under MIT (approximately 247 stars); the same code powers the hosted deployment and can be self-hosted" + "value": "Server is fully open source under MIT (approximately 259 stars, latest release v0.3.28 on 2026-07-07); the same code powers the hosted deployment and can be self-hosted" } ], "methodology": "Source code review of the published server implementation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 76, @@ -277,7 +277,7 @@ } ], "methodology": "Comparison of documented tool surface against per-user dynamic configuration", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment for hosted and local modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 80, @@ -310,7 +310,7 @@ } ], "methodology": "Latency observation across built-in and attached tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 82, @@ -324,7 +324,7 @@ } ], "methodology": "Uptime analysis of Hub infrastructure versus attached tool availability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 85, @@ -338,7 +338,7 @@ } ], "methodology": "Feature completeness assessment including the dynamic tool extension model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 72, @@ -348,11 +348,11 @@ "source": "hf-mcp-server Repository", "url": "https://github.com/huggingface/hf-mcp-server", "date": "2026-06-10", - "value": "Approximately 247 GitHub stars with active first-party maintenance; adoption driven mainly by the hosted endpoint within the large HF user base" + "value": "Approximately 259 GitHub stars with active first-party maintenance (121 releases; latest v0.3.28 on 2026-07-07); adoption driven mainly by the hosted endpoint within the large HF user base" } ], "methodology": "Community activity and adoption analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -377,12 +377,14 @@ "repository": "https://github.com/huggingface/hf-mcp-server", "license": "MIT", "maintained_by": "Hugging Face", - "github_stars": 247, + "github_stars": 259, + "latest_release": "v0.3.28 (2026-07-07)", "remote_endpoint": "https://huggingface.co/mcp", "configuration_url": "https://huggingface.co/settings/mcp", "authentication": "Hugging Face account / access tokens (fine-grained supported)", "transport_types": [ "streamable-http", + "streamable-http-json (stateless)", "stdio" ], "installation_methods": [ diff --git a/data/mcps/mcp-server-kubernetes.json b/data/mcps/mcp-server-kubernetes.json index 1eb374b..62bdaec 100644 --- a/data/mcps/mcp-server-kubernetes.json +++ b/data/mcps/mcp-server-kubernetes.json @@ -4,10 +4,10 @@ "name": "MCP Kubernetes Server", "provider": "Community", "version": "1.0.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Kubernetes cluster management. Enables AI models to interact with Kubernetes API for pod management, deployment orchestration, service configuration, and cluster resource monitoring. Essential for AI-powered Kubernetes operations and cloud-native application management.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "Community-maintained MCP servers for Kubernetes cluster management; there is still no single official upstream Kubernetes MCP server. Two implementations lead: containers/kubernetes-mcp-server (Go-native, talks directly to the Kubernetes API, supports OpenShift, Red Hat-backed, in the Red Hat Ecosystem Catalog) and Flux159/mcp-server-kubernetes (TypeScript, wraps kubectl/helm). Enables pod management, deployment orchestration, service configuration, and cluster resource monitoring.", + "website": "https://github.com/containers/kubernetes-mcp-server", "trust_vector": { "performance_reliability": { "overall_score": 84, @@ -24,7 +24,7 @@ } ], "methodology": "API stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_operation_success": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Operation success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cluster_state_accuracy": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "State synchronization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "multi_cluster_performance": { "score": 78, @@ -66,7 +66,7 @@ } ], "methodology": "Multi-cluster performance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "kubeconfig_exposure_risk": { "score": 62, @@ -113,7 +113,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "destructive_operation_risk": { "score": 58, @@ -127,7 +127,7 @@ } ], "methodology": "Operation risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "namespace_isolation": { "score": 72, @@ -141,7 +141,7 @@ } ], "methodology": "Isolation boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "secret_access_risk": { "score": 65, @@ -155,7 +155,7 @@ } ], "methodology": "Secrets management assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 82, @@ -169,7 +169,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pod_log_privacy": { "score": 63, @@ -202,7 +202,7 @@ } ], "methodology": "Log privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "secret_exposure_risk": { "score": 68, @@ -216,7 +216,7 @@ } ], "methodology": "Secret privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "configmap_data_privacy": { "score": 72, @@ -244,7 +244,7 @@ } ], "methodology": "Configuration privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 85, @@ -277,7 +277,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -288,10 +288,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Open source community implementation" + }, + { + "source": "containers/kubernetes-mcp-server", + "url": "https://github.com/containers/kubernetes-mcp-server", + "date": "2026-07-09", + "value": "Leading Go-native open-source implementation (Kubernetes and OpenShift) interacting directly with the Kubernetes API; Red Hat ships hardened container builds via its Ecosystem Catalog. Flux159/mcp-server-kubernetes (TypeScript, kubectl/helm wrapper) remains an active alternative" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "security_best_practices": { "score": 68, @@ -305,7 +311,7 @@ } ], "methodology": "Security documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +330,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_performance": { "score": 82, @@ -338,7 +344,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -352,7 +358,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_coverage": { "score": 82, @@ -366,7 +372,7 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 75, @@ -377,10 +383,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Community-maintained with moderate activity and support" + }, + { + "source": "Red Hat Developer - Kubernetes MCP server", + "url": "https://developers.redhat.com/articles/2025/09/25/kubernetes-mcp-server-ai-powered-cluster-management", + "date": "2026-07-09", + "value": "Red Hat backing of containers/kubernetes-mcp-server (including Red Hat Ecosystem Catalog distribution) has strengthened maintenance and enterprise support for the ecosystem" } ], "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -390,7 +402,8 @@ "Built on stable and mature Kubernetes API", "Excellent for cloud-native application deployment automation", "Full operation auditability through Kubernetes audit logs", - "Open source community implementation", + "Open source community implementations (Go-native containers/kubernetes-mcp-server with Red Hat backing; TypeScript Flux159/mcp-server-kubernetes)", + "OpenShift support and Red Hat Ecosystem Catalog distribution for the containers implementation", "Supports RBAC for granular access control" ], "limitations": [ @@ -399,23 +412,27 @@ "Pod logs and Kubernetes secrets accessible if RBAC permits", "Requires careful RBAC configuration to limit access", "Community-maintained with variable support quality", - "ConfigMaps and secrets may contain sensitive data" + "ConfigMaps and secrets may contain sensitive data", + "No official upstream (CNCF/Kubernetes project) MCP server; teams must choose among community implementations" ], "metadata": { - "license": "MIT", + "license": "Apache-2.0 (containers/kubernetes-mcp-server); MIT (Flux159/mcp-server-kubernetes)", "supported_platforms": [ - "All platforms with kubectl" + "Linux", + "macOS", + "Windows" ], "programming_languages": [ - "TypeScript", - "Python" + "Go (containers/kubernetes-mcp-server)", + "TypeScript (Flux159/mcp-server-kubernetes)" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", + "github_repo": "https://github.com/containers/kubernetes-mcp-server", + "alternative_repo": "https://github.com/Flux159/mcp-server-kubernetes", "api_dependency": "Kubernetes API", "authentication": "kubeconfig, Service Account tokens", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "Community (containers org with Red Hat backing; Flux159)" }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-linear.json b/data/mcps/mcp-server-linear.json index 83756b0..04d8f5d 100644 --- a/data/mcps/mcp-server-linear.json +++ b/data/mcps/mcp-server-linear.json @@ -2,12 +2,12 @@ "id": "mcp-server-linear", "type": "mcp", "name": "MCP Linear Server", - "provider": "Community", - "version": "1.0.0", - "last_evaluated": "2025-11-09", + "provider": "Linear (official; community implementations also exist)", + "version": "hosted remote", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Linear project management integration. Enables AI models to create, read, update issues, manage projects, track cycles, and coordinate teams. Essential for AI-powered software development workflows, sprint planning, and issue tracking automation.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "Linear's OFFICIAL MCP server, a hosted remote at https://mcp.linear.app/mcp using OAuth 2.1 with dynamic client registration; Bearer-token/API-key auth is supported for non-interactive or read-only access. Exposes 25+ tools to create, read, update issues, manage projects, track cycles, and coordinate teams. Supersedes earlier community-maintained Linear MCP servers, which remain available as open-source alternatives.", + "website": "https://linear.app/docs/mcp", "trust_vector": { "performance_reliability": { "overall_score": 86, @@ -24,7 +24,7 @@ } ], "methodology": "Operation success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Search quality testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "real_time_sync": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Sync reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_performance": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -96,10 +96,16 @@ "url": "https://developers.linear.app/docs/graphql/working-with-the-graphql-api#authentication", "date": "2025-11-16", "value": "Uses personal or OAuth access tokens" + }, + { + "source": "Linear Docs - MCP server", + "url": "https://linear.app/docs/mcp", + "date": "2026-07-09", + "value": "Official hosted server at https://mcp.linear.app/mcp authenticates via OAuth 2.1 with dynamic client registration (browser flow, no stored API key); Authorization: Bearer header with an OAuth token or restricted API key supported for non-interactive/read-only use" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 68, @@ -110,10 +116,16 @@ "url": "https://modelcontextprotocol.io/docs/security", "date": "2025-11-16", "value": "API token stored locally; AI can perform issue operations" + }, + { + "source": "Linear Docs - MCP server", + "url": "https://linear.app/docs/mcp", + "date": "2026-07-09", + "value": "Official hosted server removes local API-key storage for the default OAuth flow; restricted (read-only) API keys can scope access when using header auth" } ], "methodology": "Token security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "issue_modification_risk": { "score": 70, @@ -127,7 +139,7 @@ } ], "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "team_access_control": { "score": 78, @@ -141,7 +153,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "issue_deletion_risk": { "score": 72, @@ -155,7 +167,7 @@ } ], "methodology": "Deletion risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 80, @@ -169,7 +181,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +200,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "team_member_privacy": { "score": 68, @@ -202,7 +214,7 @@ } ], "methodology": "User privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "project_structure_exposure": { "score": 72, @@ -216,7 +228,7 @@ } ], "methodology": "Structure privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -230,7 +242,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "attachment_privacy": { "score": 70, @@ -244,7 +256,7 @@ } ], "methodology": "Attachment privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +275,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 85, @@ -277,7 +289,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -288,10 +300,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Open source community implementation" + }, + { + "source": "Linear Docs - MCP server", + "url": "https://linear.app/docs/mcp", + "date": "2026-07-09", + "value": "Official Linear MCP server is a closed-source hosted service; open-source community implementations remain available for users who require source transparency" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 73, @@ -305,7 +323,7 @@ } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -321,10 +339,16 @@ "url": "https://github.com/modelcontextprotocol/servers", "date": "2025-11-16", "value": "Simple setup requiring Linear API token" + }, + { + "source": "Linear Docs - MCP server", + "url": "https://linear.app/docs/mcp", + "date": "2026-07-09", + "value": "Hosted remote requires no install: add https://mcp.linear.app/mcp as an HTTP transport and complete browser OAuth" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_performance": { "score": 85, @@ -338,7 +362,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 88, @@ -352,7 +376,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage": { "score": 82, @@ -366,7 +390,7 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 75, @@ -377,10 +401,16 @@ "url": "https://github.com/modelcontextprotocol/servers/discussions", "date": "2025-11-16", "value": "Community-maintained with moderate activity" + }, + { + "source": "Linear Docs - MCP server", + "url": "https://linear.app/docs/mcp", + "date": "2026-07-09", + "value": "Now officially maintained by Linear as a first-party hosted product with vendor documentation and support" } ], "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -390,7 +420,8 @@ "Built on fast and reliable GraphQL API", "Excellent for agile development and sprint planning workflows", "Comprehensive activity logging and audit trail", - "Open source community implementation", + "Official Linear-maintained hosted remote (https://mcp.linear.app/mcp) with OAuth 2.1 dynamic client registration; no local API-key storage needed in the default flow, restricted read-only API keys supported", + "Open source community implementations available as alternatives", "Real-time updates via webhooks and subscriptions" ], "limitations": [ @@ -399,23 +430,29 @@ "Team member information and activity data accessible", "Project roadmaps and sprint planning data may be revealed", "Issue attachments and links accessible", - "Requires careful team permission configuration" + "Requires careful team permission configuration", + "Official hosted server is a closed-source managed service (use community open-source servers if source transparency is required)" ], "metadata": { - "license": "MIT", + "license": "Proprietary (hosted service); community implementations MIT", "supported_platforms": [ - "All platforms with Node.js/Python" + "Hosted remote (https://mcp.linear.app/mcp)", + "Any MCP client with HTTP transport" ], "programming_languages": [ - "TypeScript", - "Python" + "N/A (hosted service)" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", + "docs": "https://linear.app/docs/mcp", "api_dependency": "Linear GraphQL API", - "authentication": "Linear API Token or OAuth", - "first_release": "2024-11", - "maintained_by": "Community" + "authentication": "OAuth 2.1 with dynamic client registration; Bearer token / API key header supported", + "remote_endpoint": "https://mcp.linear.app/mcp", + "first_release": "2024-11 (community); 2025-05 (official hosted)", + "maintained_by": "Linear", + "status": "Active - official Linear hosted server; community servers remain as open-source alternatives", + "transport_types": [ + "streamable-http" + ] }, "use_case_ratings": { "code-generation": { @@ -468,6 +505,8 @@ "project-management", "issue-tracking", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "remote-server" ] } diff --git a/data/mcps/mcp-server-memory.json b/data/mcps/mcp-server-memory.json index a28d4d2..3edc8f4 100644 --- a/data/mcps/mcp-server-memory.json +++ b/data/mcps/mcp-server-memory.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Memory Server", "provider": "Anthropic", - "version": "2025.9.25", - "last_evaluated": "2026-06-10", + "version": "2026.7.4", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "MCP reference server providing persistent memory and knowledge graph capabilities: long-term retention, entity relationship tracking, and contextual recall across conversations. One of the seven reference servers still actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest spec 2025-11-25. Critical for personalized AI but raises significant privacy concerns.", + "description": "MCP reference server providing persistent memory and knowledge graph capabilities: long-term retention, entity relationship tracking, and contextual recall across conversations. One of the seven reference servers still maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest stable spec 2025-11-25 (2026-07-28 revision at release-candidate stage). Critical for personalized AI but raises privacy concerns.", "website": "https://github.com/modelcontextprotocol/servers", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Retrieval accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "knowledge_graph_integrity": { "score": 82, @@ -38,7 +38,7 @@ } ], "methodology": "Graph consistency testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "context_recall": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Recall quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "storage_scalability": { "score": 84, @@ -66,7 +66,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_performance": { "score": 81, @@ -80,7 +80,7 @@ } ], "methodology": "Query latency testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_modification_risk": { "score": 70, @@ -113,7 +113,7 @@ } ], "methodology": "Write operation risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "memory_poisoning_risk": { "score": 68, @@ -127,7 +127,7 @@ } ], "methodology": "Data integrity risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_injection_protection": { "score": 78, @@ -141,7 +141,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 72, @@ -155,7 +155,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Privacy risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_retention_control": { "score": 68, @@ -188,7 +188,7 @@ } ], "methodology": "Data retention controls review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "consent_management": { "score": 62, @@ -202,7 +202,7 @@ } ], "methodology": "Consent framework review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "right_to_deletion": { "score": 70, @@ -216,7 +216,7 @@ } ], "methodology": "GDPR right to erasure assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "embedding_privacy": { "score": 65, @@ -230,7 +230,7 @@ } ], "methodology": "Embedding privacy analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Memory transparency assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "knowledge_graph_explainability": { "score": 75, @@ -263,7 +263,7 @@ } ], "methodology": "Explainability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "documentation_quality": { "score": 78, @@ -277,7 +277,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 80, @@ -291,7 +291,7 @@ } ], "methodology": "Source code transparency review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -310,7 +310,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "storage_efficiency": { "score": 78, @@ -324,7 +324,7 @@ } ], "methodology": "Storage efficiency testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "retrieval_performance": { "score": 81, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "maintenance_requirements": { "score": 77, @@ -355,10 +355,16 @@ "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", "date": "2025-12-09", "value": "memory is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" + }, + { + "source": "npm - @modelcontextprotocol/server-memory", + "url": "https://www.npmjs.com/package/@modelcontextprotocol/server-memory", + "date": "2026-07-09", + "value": "Latest release 2026.7.4 (published 2026-07-04); still listed among the seven maintained reference servers in modelcontextprotocol/servers README (repo ~88,300 stars, not archived)" } ], "methodology": "Maintenance overhead assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "backup_and_recovery": { "score": 80, @@ -372,7 +378,7 @@ } ], "methodology": "Backup capabilities review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -392,7 +398,7 @@ "Limited consent management and data retention controls", "Memory poisoning risk - AI can store incorrect information", "GDPR/privacy compliance challenges without careful implementation", - "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" + "STATUS 2026-07-09: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); latest npm release 2026.7.4 (2026-07-04); no known CVEs against this server; governance under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -405,7 +411,7 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/modelcontextprotocol/servers", - "github_stars": 58700, + "github_stars": 88259, "storage_backends": [ "SQLite", "PostgreSQL", diff --git a/data/mcps/mcp-server-mongodb.json b/data/mcps/mcp-server-mongodb.json index a221693..8b73865 100644 --- a/data/mcps/mcp-server-mongodb.json +++ b/data/mcps/mcp-server-mongodb.json @@ -3,11 +3,11 @@ "type": "mcp", "name": "MCP MongoDB Server", "provider": "MongoDB (Official)", - "version": "1.0.0", - "last_evaluated": "2025-11-09", + "version": "1.13.0", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for MongoDB database operations. Enables AI models to query, insert, update, and delete documents, manage collections, create indexes, and perform aggregations. Essential for AI-powered NoSQL database management and data analysis workflows.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "Official MongoDB MCP server (mongodb-js/mongodb-mcp-server), generally available since v1.9.0 (2026-03-24). Enables AI models to query, insert, update, and delete documents, manage collections and indexes, run aggregations, and administer MongoDB Atlas clusters (50+ tools spanning database, Atlas, Atlas Local, and Assistant tools). Supports a read-only mode, elicitation-based user confirmation for destructive operations, and disables server-side JavaScript by default.", + "website": "https://github.com/mongodb-js/mongodb-mcp-server", "trust_vector": { "performance_reliability": { "overall_score": 83, @@ -24,7 +24,7 @@ } ], "methodology": "Query success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "aggregation_accuracy": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Aggregation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "large_dataset_handling": { "score": 78, @@ -46,13 +46,13 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", "value": "Performance degrades with very large result sets; pagination needed" } ], "methodology": "Scalability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "connection_stability": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Connection stability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 82, @@ -74,18 +74,18 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", "value": "Handles database errors with retry logic and error reporting" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 68, + "overall_score": 71, "criteria": { "authentication_security": { "score": 78, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "connection_string_exposure": { "score": 60, @@ -113,7 +113,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_injection_risk": { "score": 65, @@ -121,41 +121,41 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", "value": "AI can construct arbitrary queries; NoSQL injection risk if not validated" } ], "methodology": "Injection vulnerability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_modification_control": { - "score": 62, + "score": 70, "confidence": "high", "evidence": [ { - "source": "MongoDB Permissions", - "url": "https://www.mongodb.com/docs/manual/core/authorization/", - "date": "2025-11-16", - "value": "AI can insert, update, and delete documents within user permissions" + "source": "MongoDB MCP Server Repository", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", + "value": "AI can insert, update, and delete documents within user permissions, but the official server offers --readOnly / MDB_MCP_READ_ONLY=true to restrict tools to read, connect, and metadata operation types" } ], - "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "methodology": "Operation authorization testing; raised from 62 due to the documented read-only mode in the official server", + "last_verified": "2026-07-09" }, "database_drop_risk": { - "score": 70, - "confidence": "medium", + "score": 76, + "confidence": "high", "evidence": [ { - "source": "Implementation Review", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Can drop collections and databases if permissions allow" + "source": "MongoDB MCP Server Winter 2026 Release Notes", + "url": "https://www.mongodb.com/company/blog/product-release-announcements/whats-new-mongodb-mcp-server-winter-2026-edition", + "date": "2026-07-09", + "value": "Destructive operations (drop-database, drop-collection, delete-many) now use MCP elicitation to prompt for explicit user confirmation before executing" } ], - "methodology": "Destructive operation testing", - "last_verified": "2025-11-09" + "methodology": "Destructive operation testing; raised from 70 following elicitation-based confirmation added in the GA releases", + "last_verified": "2026-07-09" }, "audit_logging": { "score": 75, @@ -169,7 +169,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 58, @@ -196,13 +196,13 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", "value": "No built-in PII detection or filtering; all queried data exposed" } ], "methodology": "PII protection assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "field_level_security": { "score": 70, @@ -216,7 +216,7 @@ } ], "methodology": "Field security assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_pattern_exposure": { "score": 72, @@ -238,32 +238,32 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", "value": "Query patterns may reveal data structure and business logic" } ], "methodology": "Query privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "trust_transparency": { - "overall_score": 78, + "overall_score": 80, "criteria": { "documentation_quality": { - "score": 75, - "confidence": "medium", + "score": 84, + "confidence": "high", "evidence": [ { - "source": "MongoDB MCP Docs", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Good documentation but community-maintained with evolving coverage" + "source": "MongoDB MCP Server Documentation", + "url": "https://www.mongodb.com/docs/mcp-server/", + "date": "2026-07-09", + "value": "Official MongoDB documentation site with get-started guides, tool reference, and configuration options, maintained alongside the GA product" } ], - "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "methodology": "Documentation completeness review; raised from 75 as official first-party docs replaced community-maintained coverage", + "last_verified": "2026-07-09" }, "query_visibility": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Query logging assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -285,13 +285,13 @@ "evidence": [ { "source": "GitHub Repository", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Open source community implementation" + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", + "value": "Open source under Apache 2.0, officially maintained by MongoDB (1.1k+ stars)" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage_clarity": { "score": 70, @@ -299,18 +299,18 @@ "evidence": [ { "source": "MCP Server Documentation", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", "value": "Clear but incomplete documentation of supported MongoDB operations" } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 80, + "overall_score": 82, "criteria": { "ease_of_setup": { "score": 82, @@ -318,13 +318,13 @@ "evidence": [ { "source": "Setup Documentation", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", "value": "Simple setup requiring MongoDB connection string" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_performance": { "score": 78, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 82, @@ -352,70 +352,73 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage": { - "score": 80, + "score": 88, "confidence": "high", "evidence": [ { - "source": "MongoDB MCP Server", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Covers CRUD operations, aggregations, indexing, and collection management" + "source": "MongoDB MCP Server Repository", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", + "value": "50+ tools: CRUD, aggregations, indexing, and collection management plus Atlas cluster/user management, Atlas Local tools, GA lexical and vector search, and sample dataset loading" } ], - "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "methodology": "Feature coverage assessment; raised from 80 reflecting the expanded GA tool surface including Atlas management", + "last_verified": "2026-07-09" }, "community_support": { - "score": 75, - "confidence": "medium", + "score": 82, + "confidence": "high", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/modelcontextprotocol/servers/discussions", - "date": "2025-11-16", - "value": "Community-maintained with moderate activity" + "source": "MongoDB MCP Server Repository", + "url": "https://github.com/mongodb-js/mongodb-mcp-server", + "date": "2026-07-09", + "value": "Officially maintained by MongoDB with frequent releases (latest stable v1.13.0, 2026-06-16), 1.1k+ stars, and a companion mongodb/agent-skills package" } ], - "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "methodology": "Community support assessment; raised from 75 reflecting first-party GA maintenance", + "last_verified": "2026-07-09" } } } }, "strengths": [ - "Comprehensive MongoDB CRUD and aggregation capabilities", - "Built on official MongoDB drivers with stable performance", - "Excellent for NoSQL data analysis and document processing", - "Supports complex aggregation pipelines and queries", - "Open source community implementation", - "Flexible schema-less data handling" + "Official MongoDB implementation, GA since March 2026, with 50+ tools", + "Comprehensive CRUD, aggregation, and Atlas cluster management capabilities", + "Read-only mode (--readOnly) and elicitation-based confirmation for destructive operations", + "Server-side JavaScript disabled by default as a security measure", + "GA lexical and vector search tools for AI retrieval workloads", + "Open source (Apache 2.0) with first-party MongoDB maintenance and official docs" ], "limitations": [ "Query results and document data exposed to LLM provider", "No built-in PII detection or sensitive data filtering", "Risk of NoSQL injection without proper query validation", - "AI can modify or delete data within permission scope", - "Connection strings with credentials accessible to AI", + "AI can modify or delete data within permission scope unless read-only mode is set", + "Connection strings with credentials stored in local configuration", "Limited audit logging in MongoDB Community Edition" ], "metadata": { - "license": "MIT", + "license": "Apache 2.0", "supported_platforms": [ - "All platforms with Node.js/Python" + "All platforms with Node.js; Docker" ], "programming_languages": [ - "TypeScript", - "Python" + "TypeScript" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", - "api_dependency": "MongoDB Node.js Driver, PyMongo", - "authentication": "MongoDB connection string (username/password)", - "first_release": "2024-11", - "maintained_by": "Community" + "github_repo": "https://github.com/mongodb-js/mongodb-mcp-server", + "github_stars": 1100, + "docs": "https://www.mongodb.com/docs/mcp-server/", + "api_dependency": "MongoDB Node.js Driver, Atlas Administration API", + "authentication": "MongoDB connection string (MDB_MCP_CONNECTION_STRING) or Atlas API service account (client ID/secret)", + "security_controls": ["read-only mode (--readOnly / MDB_MCP_READ_ONLY)", "elicitation confirmation for destructive operations", "server-side JavaScript disabled by default"], + "ga_date": "2026-03-24", + "first_release": "2025 (public preview)", + "maintained_by": "MongoDB" }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-notion.json b/data/mcps/mcp-server-notion.json index ea0475a..26d3a82 100644 --- a/data/mcps/mcp-server-notion.json +++ b/data/mcps/mcp-server-notion.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Notion Server", "provider": "Notion (Official)", - "version": "1.0.0", - "last_evaluated": "2026-06-10", + "version": "2.4.1", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Notion's official MCP integration. Primary offering is the hosted MCP server at https://mcp.notion.com/mcp (Streamable HTTP and SSE transports, one-click OAuth), positioned as the successor to the open-source local @notionhq/notion-mcp-server (MIT, 4,407 stars), which remains available. Enables AI models to create, read, update pages and databases, manage blocks, and search workspace content.", + "description": "Notion's official MCP integration. Primary offering is the hosted server at https://mcp.notion.com/mcp (Streamable HTTP and SSE, one-click OAuth), the only actively supported path. The open-source local @notionhq/notion-mcp-server (MIT, v2.4.1) is in maintenance mode: issues/PRs not actively monitored and Notion may sunset the repo. Enables creating, reading, updating pages and databases, managing blocks, and searching workspaces; recent releases track Notion API version 2026-03-11.", "website": "https://mcp.notion.com/mcp", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Operation success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 82, @@ -38,7 +38,7 @@ } ], "methodology": "Search quality testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 75, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "block_rendering_accuracy": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Content rendering testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -105,7 +105,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 65, @@ -119,7 +119,7 @@ } ], "methodology": "Token security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "page_modification_risk": { "score": 68, @@ -133,7 +133,7 @@ } ], "methodology": "Operation authorization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "workspace_access_control": { "score": 75, @@ -147,7 +147,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "content_deletion_risk": { "score": 70, @@ -161,7 +161,7 @@ } ], "methodology": "Deletion risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 78, @@ -175,7 +175,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -194,7 +194,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "database_structure_exposure": { "score": 65, @@ -208,7 +208,7 @@ } ], "methodology": "Database privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "workspace_metadata_privacy": { "score": 70, @@ -222,7 +222,7 @@ } ], "methodology": "Metadata privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -236,7 +236,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "collaborator_information": { "score": 70, @@ -250,7 +250,7 @@ } ], "methodology": "Collaborator privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -269,7 +269,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -283,7 +283,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -292,12 +292,12 @@ { "source": "Notion MCP Server GitHub Repository", "url": "https://github.com/makenotion/notion-mcp-server", - "date": "2026-06-10", - "value": "Local server (@notionhq/notion-mcp-server) is open source under MIT with 4,407 GitHub stars; hosted server internals documented in Notion's engineering blog" + "date": "2026-07-09", + "value": "Local server (@notionhq/notion-mcp-server, v2.4.1) is open source under MIT with 4,501 GitHub stars, but the README states it is no longer actively maintained (issues/PRs not monitored, possible future sunset); hosted server internals documented in Notion's engineering blog" } ], "methodology": "Source code review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 70, @@ -311,7 +311,7 @@ } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -336,7 +336,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 78, @@ -350,7 +350,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -364,7 +364,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage": { "score": 82, @@ -378,7 +378,7 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 80, @@ -387,12 +387,12 @@ { "source": "Notion MCP Server GitHub Repository", "url": "https://github.com/makenotion/notion-mcp-server", - "date": "2026-06-10", - "value": "Officially maintained by Notion; open-source local server has 4,407 GitHub stars, and the hosted server is operated by Notion as the primary offering" + "date": "2026-07-09", + "value": "Hosted server is operated by Notion as the only actively supported offering; open-source local server (4,501 GitHub stars, last release v2.4.1 on 2026-06-22) is in maintenance mode with issues/PRs not actively monitored" } ], "methodology": "Community support assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -402,17 +402,17 @@ "Built on reliable Notion API with good uptime", "Excellent for knowledge management and documentation workflows", "Integration permissions provide page-level access control", - "Official hosted server (mcp.notion.com/mcp) with one-click OAuth; open-source local server (MIT, 4,407 stars) still available", + "Official hosted server (mcp.notion.com/mcp) with one-click OAuth; open-source local server (MIT, 4,501 stars) still installable", "Archived pages are recoverable (not permanently deleted)" ], "limitations": [ - "Page content and database data exposed to LLM provider", + "Page content, database data, schemas, and property values exposed to LLM provider", "AI can modify and archive pages within integration permissions", "Workspace structure and page hierarchy may be revealed", "Subject to strict rate limits (3 requests/second average)", - "Database schemas and property values exposed", "User and collaborator information accessible", - "STATUS 2026-06-10: Notion's hosted server (mcp.notion.com/mcp) is now the primary, recommended offering; the local @notionhq/notion-mcp-server remains available but workspace traffic on the hosted path routes through Notion's infrastructure" + "STATUS 2026-07-09: Notion's hosted server (mcp.notion.com/mcp) is the only actively supported offering; the local @notionhq/notion-mcp-server repo is no longer actively maintained (issues/PRs unmonitored, sunset possible), and workspace traffic on the hosted path routes through Notion's infrastructure", + "Notion API version 2026-03-11 introduced breaking changes (after→position, archived→in_trash, transcription→meeting_notes) that older local-server versions do not handle" ], "metadata": { "license": "MIT", @@ -425,7 +425,8 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/makenotion/notion-mcp-server", - "github_stars": 4407, + "github_stars": 4501, + "local_server_status": "Maintenance mode since 2026: no longer actively maintained, may be sunset", "api_dependency": "Notion API, Notion Client SDK", "authentication": "OAuth (hosted at mcp.notion.com/mcp) or Notion Integration Token (local server)", "remote_endpoint": "https://mcp.notion.com/mcp", diff --git a/data/mcps/mcp-server-perplexity.json b/data/mcps/mcp-server-perplexity.json index 6a38a3a..f943686 100644 --- a/data/mcps/mcp-server-perplexity.json +++ b/data/mcps/mcp-server-perplexity.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Perplexity Server", "provider": "Perplexity AI", - "version": "2025.1.0", - "last_evaluated": "2025-01-14", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "MCP server enabling AI models to leverage Perplexity's advanced search capabilities. Provides multi-source research with automatic citations, recency filtering, and domain-specific search modes for comprehensive information retrieval.", + "description": "Perplexity's official MCP server (perplexityai/modelcontextprotocol, npm @perplexity-ai/mcp-server) for the Perplexity API Platform. Exposes four tools — perplexity_search (Search API), perplexity_ask (sonar-pro), perplexity_research (sonar-deep-research), and perplexity_reason (sonar-reasoning-pro) — providing multi-source research with automatic citations and recency filtering for comprehensive information retrieval.", "website": "https://www.perplexity.ai/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Search quality testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "citation_quality": { "score": 95, @@ -38,7 +38,7 @@ } ], "methodology": "Citation accuracy testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "response_quality": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Response quality assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "recency_filtering": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Recency testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "api_reliability": { "score": 85, @@ -80,7 +80,7 @@ } ], "methodology": "Reliability testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_handling": { "score": 78, @@ -113,7 +113,7 @@ } ], "methodology": "Data handling review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "content_safety": { "score": 82, @@ -127,7 +127,7 @@ } ], "methodology": "Content safety assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "enterprise_security": { "score": 85, @@ -141,7 +141,7 @@ } ], "methodology": "Enterprise security review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -160,7 +160,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "enterprise_privacy": { "score": 85, @@ -174,7 +174,7 @@ } ], "methodology": "Enterprise privacy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "score": 78, @@ -188,7 +188,7 @@ } ], "methodology": "Retention policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -207,7 +207,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "source_attribution": { "score": 95, @@ -221,7 +221,7 @@ } ], "methodology": "Attribution testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "model_transparency": { "score": 82, @@ -235,21 +235,21 @@ } ], "methodology": "Model transparency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "pricing_clarity": { "score": 88, "confidence": "high", "evidence": [ { - "source": "Perplexity Pricing", - "url": "https://www.perplexity.ai/pricing", - "date": "2025-01-10", - "value": "Clear pricing with defined API quotas" + "source": "Perplexity API Pricing", + "url": "https://docs.perplexity.ai/getting-started/pricing", + "date": "2026-07-09", + "value": "Published per-token pricing for all Sonar models: sonar $1/$1 per 1M tokens (in/out), sonar-pro $3/$15, sonar-reasoning-pro $2/$8, sonar-deep-research $2/$8 plus citation ($2/1M), reasoning ($3/1M), and search-query ($5/1K) fees; per-request search fees scale with context size ($5-$12 per 1K requests for sonar, $6-$14 for sonar-pro)" } ], "methodology": "Pricing review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -268,7 +268,7 @@ } ], "methodology": "Setup assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "response_speed": { "score": 85, @@ -282,49 +282,49 @@ } ], "methodology": "Performance testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "model_selection": { "score": 88, "confidence": "high", "evidence": [ { - "source": "Perplexity Models", + "source": "Perplexity API Platform docs", "url": "https://docs.perplexity.ai/", - "date": "2025-01-10", - "value": "Multiple model tiers for different use cases" + "date": "2026-07-09", + "value": "Current Sonar model lineup spans sonar, sonar-pro, sonar-reasoning-pro, and sonar-deep-research, each mapped to a dedicated MCP tool for search, ask, reason, and research workflows" } ], "methodology": "Model options review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "cost_efficiency": { "score": 82, "confidence": "medium", "evidence": [ { - "source": "Perplexity Pricing", - "url": "https://www.perplexity.ai/pricing", - "date": "2025-01-10", - "value": "Competitive pricing; can be cost-effective vs. multiple API calls" + "source": "Perplexity API Pricing", + "url": "https://docs.perplexity.ai/getting-started/pricing", + "date": "2026-07-09", + "value": "Base sonar at $1/$1 per 1M tokens is competitive, but per-request search fees ($5-$14 per 1K requests depending on model and context size) add up for high-volume agents; deep-research calls carry additional citation/reasoning/search-query fees" } ], "methodology": "Cost analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "integration_quality": { "score": 88, "confidence": "high", "evidence": [ { - "source": "Perplexity SDK", - "url": "https://docs.perplexity.ai/", - "date": "2025-01-10", - "value": "Good SDK support with MCP integration" + "source": "Perplexity MCP Server guide", + "url": "https://docs.perplexity.ai/guides/mcp-server", + "date": "2026-07-09", + "value": "Official MCP server (@perplexity-ai/mcp-server v0.9.0) installed via npx with one-click installers for Cursor and VS Code; stdio transport with PERPLEXITY_API_KEY env var" } ], "methodology": "Integration testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } @@ -340,23 +340,26 @@ "limitations": [ "Queries processed through Perplexity servers", "Consumer tier data may be used for training", - "No self-hosted option available", - "Costs can accumulate with heavy usage", + "No self-hosted option for the underlying API", + "Costs can accumulate with heavy usage — per-request search fees stack on top of token pricing", "Rate limits based on pricing tier", - "Dependent on Perplexity's infrastructure" + "Local stdio server only — no hosted remote MCP endpoint documented" ], "metadata": { - "license": "Proprietary (API Service)", - "supported_platforms": ["All platforms with HTTP"], - "programming_languages": ["Python", "JavaScript", "TypeScript"], + "license": "MIT (MCP server); proprietary hosted API service", + "supported_platforms": ["All platforms with Node.js"], + "programming_languages": ["TypeScript"], "mcp_version": "1.0", "website": "https://www.perplexity.ai/", - "api_dependency": "Perplexity API", - "authentication": "Bearer Token", + "github_repo": "https://github.com/perplexityai/modelcontextprotocol", + "package": "@perplexity-ai/mcp-server", + "package_version": "0.9.0", + "api_dependency": "Perplexity API Platform (Sonar models + Search API)", + "authentication": "Bearer Token (PERPLEXITY_API_KEY)", "first_release": "2024", "maintained_by": "Perplexity AI", "transport_types": ["stdio"], - "installation_methods": ["npm", "pip"] + "installation_methods": ["npm", "npx"] }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-playwright.json b/data/mcps/mcp-server-playwright.json index 03b7b74..1181ece 100644 --- a/data/mcps/mcp-server-playwright.json +++ b/data/mcps/mcp-server-playwright.json @@ -3,8 +3,8 @@ "type": "mcp", "name": "Playwright MCP", "provider": "Microsoft", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Microsoft's official MCP server for browser automation via Playwright. Exposes 25+ tools (navigation, clicking, typing, form filling, screenshots, network inspection, JS evaluation) that operate on structured accessibility-tree snapshots rather than pixels, making agent-driven browsing fast and deterministic. Supersedes the archived Puppeteer reference server.", "website": "https://github.com/microsoft/playwright-mcp", @@ -24,7 +24,7 @@ } ], "methodology": "Review of snapshot mechanism (browser_snapshot) and element-reference stability across page interactions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 88, @@ -38,7 +38,7 @@ } ], "methodology": "Hands-on testing of navigation, clicking, typing, and form-fill tools against common web applications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "browser_compatibility": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Cross-browser capability review based on underlying Playwright engine support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 84, @@ -66,7 +66,7 @@ } ], "methodology": "Error-path testing including timeouts, missing elements, modal dialogs, and navigation failures", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "automation_stability": { "score": 86, @@ -80,7 +80,7 @@ } ], "methodology": "Multi-step workflow stability testing across tabs, dialogs, and dynamic pages", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Threat modeling of untrusted web content entering the agent context via snapshots and screenshots", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "arbitrary_code_execution_risk": { "score": 50, @@ -113,7 +113,7 @@ } ], "methodology": "Capability analysis of the browser_evaluate tool and its abuse potential under prompt injection", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sandboxing_isolation": { "score": 75, @@ -127,7 +127,7 @@ } ], "methodology": "Review of --isolated, headless, and origin-filtering configuration flags as mitigations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 62, @@ -141,7 +141,7 @@ } ], "methodology": "Analysis of session/cookie access in persistent vs isolated profile modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 60, @@ -155,7 +155,7 @@ } ], "methodology": "Authorization boundary analysis of write-capable browsing actions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of snapshot, screenshot, and network tool outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 62, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy controls assessment of snapshot and screenshot content handling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_data_control": { "score": 80, @@ -202,7 +202,7 @@ } ], "methodology": "Review of local execution model and data residency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing pathway analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -249,7 +249,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -263,7 +263,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "vendor_credibility": { "score": 94, @@ -272,12 +272,12 @@ { "source": "GitHub API", "url": "https://api.github.com/repos/microsoft/playwright-mcp", - "date": "2026-06-10", - "value": "Maintained by Microsoft's Playwright team; 33,734 GitHub stars as of 2026-06-10, one of the most adopted MCP servers" + "date": "2026-07-09", + "value": "Maintained by Microsoft's Playwright team; 34,870 GitHub stars as of 2026-07-09, one of the most adopted MCP servers" } ], "methodology": "Maintainer reputation and project health analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment across MCP hosts", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "performance": { "score": 85, @@ -310,7 +310,7 @@ } ], "methodology": "Latency and token-efficiency comparison against pixel-based browser automation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 90, @@ -324,7 +324,7 @@ } ], "methodology": "Feature completeness assessment against common browser-automation needs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 92, @@ -333,12 +333,12 @@ { "source": "GitHub API", "url": "https://api.github.com/repos/microsoft/playwright-mcp", - "date": "2026-06-10", - "value": "33,734 stars; bundled or recommended by major MCP hosts and widely used as the default browser-automation server, replacing the archived Puppeteer reference server" + "date": "2026-07-09", + "value": "34,870 stars as of 2026-07-09; bundled or recommended by major MCP hosts and widely used as the default browser-automation server, replacing the archived Puppeteer reference server" } ], "methodology": "Adoption metrics and ecosystem-integration analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "maintenance_activity": { "score": 93, @@ -347,12 +347,12 @@ { "source": "GitHub repository activity", "url": "https://github.com/microsoft/playwright-mcp/commits/main", - "date": "2026-06-10", - "value": "Frequent releases tracking Playwright versions, active issue triage by the Microsoft Playwright team" + "date": "2026-07-09", + "value": "Frequent releases tracking Playwright versions (@playwright/mcp at v0.0.77 on npm, repo active through late June 2026), active issue triage by the Microsoft Playwright team" } ], "methodology": "Commit frequency and release-cadence analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -360,7 +360,7 @@ "strengths": [ "Accessibility-tree snapshots give fast, deterministic, token-efficient page interaction without vision models", "Comprehensive tool set: navigation, forms, screenshots, network inspection, tabs, dialogs, file upload", - "Backed and actively maintained by Microsoft's Playwright team (33.7k stars)", + "Backed and actively maintained by Microsoft's Playwright team (34.9k stars)", "Cross-browser support (Chromium, Firefox, WebKit) with Playwright's auto-waiting reliability", "Strong mitigation options: isolated profiles, headless mode, origin allow/block lists", "Flexible transports: stdio by default plus standalone HTTP/SSE server mode" @@ -383,8 +383,9 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/microsoft/playwright-mcp", - "github_stars": 33734, + "github_stars": 34870, "package": "@playwright/mcp", + "package_version": "0.0.77", "api_dependency": "Playwright (Chromium/Firefox/WebKit)", "authentication": "None required (local browser control)", "first_release": "2025-03", diff --git a/data/mcps/mcp-server-postgres.json b/data/mcps/mcp-server-postgres.json index 01ad894..d21a0ff 100644 --- a/data/mcps/mcp-server-postgres.json +++ b/data/mcps/mcp-server-postgres.json @@ -4,9 +4,9 @@ "name": "MCP PostgreSQL Server", "provider": "Anthropic (Archived)", "version": "0.6.2", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED WITH UNPATCHED VULNERABILITY: Former Anthropic reference MCP server for PostgreSQL, archived 2025-05-29. A SQL injection flaw disclosed by Trend Micro (June 2025) remains unpatched because the repo is archived, yet the npm package still saw ~21k weekly downloads. NOT RECOMMENDED for any use; a patched community fork (@zeddotdev/postgres-context-server) exists.", + "description": "ARCHIVED WITH UNPATCHED VULNERABILITY: Former Anthropic reference MCP server for PostgreSQL, archived 2025-05-29. A SQL injection flaw disclosed by Trend Micro (June 2025) remains unpatched because the repo is archived; the npm package (v0.6.2) is still published and downloaded. NOT RECOMMENDED for any use; a patched community fork (@zeddotdev/postgres-context-server, fixed in v0.1.4) and maintained alternatives (e.g., Microsoft's MCP server for Azure Database for PostgreSQL) exist.", "website": "https://modelcontextprotocol.io/docs/servers/postgres", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Query execution testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "schema_introspection": { "score": 92, @@ -38,7 +38,7 @@ } ], "methodology": "Schema discovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "connection_stability": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Connection reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "large_result_handling": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Large dataset performance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 86, @@ -80,7 +80,7 @@ } ], "methodology": "Error scenario testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -102,10 +102,16 @@ "url": "https://securitylabs.datadoghq.com/articles/mcp-vulnerability-case-study-SQL-injection-in-the-postgresql-mcp-server/", "date": "2025-06-30", "value": "Case study confirming SQL injection in the PostgreSQL MCP server allows bypassing the read-only transaction restriction; ~21k weekly npm downloads while vulnerable" + }, + { + "source": "npm: @modelcontextprotocol/server-postgres", + "url": "https://www.npmjs.com/package/@modelcontextprotocol/server-postgres", + "date": "2026-07-09", + "value": "v0.6.2 remains the latest published version as of July 2026; SQL injection still unpatched, no CVE identifier assigned. Patched fork @zeddotdev/postgres-context-server fixed the flaw in v0.1.4" } ], "methodology": "Vulnerability disclosure review and exploitability analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "access_control": { "score": 55, @@ -125,7 +131,7 @@ } ], "methodology": "Permission boundary testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_modification_risk": { "score": 45, @@ -145,7 +151,7 @@ } ], "methodology": "Write operation risk assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_security": { "score": 72, @@ -159,7 +165,7 @@ } ], "methodology": "Credential storage review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 68, @@ -173,7 +179,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -192,7 +198,7 @@ } ], "methodology": "Data flow and exposure analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 60, @@ -206,7 +212,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_result_filtering": { "score": 70, @@ -220,7 +226,7 @@ } ], "methodology": "Data filtering capabilities review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance_readiness": { "score": 65, @@ -234,7 +240,7 @@ } ], "methodology": "Compliance framework review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "schema_exposure": { "score": 75, @@ -248,7 +254,7 @@ } ], "methodology": "Metadata exposure analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -267,7 +273,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_visibility": { "score": 90, @@ -281,7 +287,7 @@ } ], "methodology": "Query traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 95, @@ -295,7 +301,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "security_documentation": { "score": 50, @@ -315,7 +321,7 @@ } ], "methodology": "Security documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -334,7 +340,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "performance": { "score": 84, @@ -348,7 +354,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "connection_pooling": { "score": 82, @@ -362,7 +368,7 @@ } ], "methodology": "Connection management review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_diagnostics": { "score": 80, @@ -376,7 +382,7 @@ } ], "methodology": "Error messaging assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 55, @@ -390,7 +396,7 @@ } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -410,7 +416,7 @@ "Limited granular access control beyond database user permissions", "Query results with sensitive data sent to external APIs", "Compliance challenges for regulated industries (HIPAA, PCI-DSS, GDPR)", - "ARCHIVED 2025-05-29 with an UNPATCHED SQL injection vulnerability (Trend Micro, June 2025); use the patched fork @zeddotdev/postgres-context-server instead" + "ARCHIVED 2025-05-29 with an UNPATCHED SQL injection vulnerability (Trend Micro, June 2025) confirmed still unpatched in npm v0.6.2 as of July 2026; use the patched fork @zeddotdev/postgres-context-server (v0.1.4+) or a maintained alternative instead" ], "metadata": { "license": "MIT", @@ -428,7 +434,7 @@ "connection_method": "Connection string with credentials", "first_release": "2024-11", "maintained_by": "None (Archived 2025-05-29)", - "status": "Archived - unpatched SQL injection vulnerability; patched community fork: @zeddotdev/postgres-context-server", + "status": "Archived - unpatched SQL injection vulnerability (still unpatched as of 2026-07); patched community fork: @zeddotdev/postgres-context-server (v0.1.4+); maintained alternative: Microsoft MCP server for Azure Database for PostgreSQL", "package_name": "@modelcontextprotocol/server-postgres", "transport_types": [ "stdio" diff --git a/data/mcps/mcp-server-puppeteer.json b/data/mcps/mcp-server-puppeteer.json index 7ba9912..5925e08 100644 --- a/data/mcps/mcp-server-puppeteer.json +++ b/data/mcps/mcp-server-puppeteer.json @@ -4,9 +4,9 @@ "name": "MCP Puppeteer Server", "provider": "Anthropic (Archived)", "version": "2025.4.6", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former Anthropic reference MCP server for Puppeteer browser automation, archived 2025-05-29 to the servers-archived repository and no longer maintained. The archived repo explicitly provides no security guarantees. For browser automation, Microsoft's Playwright MCP server is the recommended successor.", + "description": "ARCHIVED: Former Anthropic reference MCP server for Puppeteer browser automation, archived 2025-05-29 to the servers-archived repository and no longer maintained. The archived repo explicitly provides no security guarantees. For browser automation, Microsoft's Playwright MCP server (microsoft/playwright-mcp) is the recommended successor and remains actively maintained as of July 2026.", "website": "https://pptr.dev/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Automation task success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "page_load_reliability": { "score": 82, @@ -38,7 +38,7 @@ } ], "methodology": "Page load success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "element_interaction": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Element interaction testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "screenshot_quality": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Screenshot quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 75, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_usage": { "score": 72, @@ -94,7 +94,7 @@ } ], "methodology": "Resource utilization monitoring", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -113,7 +113,7 @@ } ], "methodology": "Sandbox security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "arbitrary_code_execution_risk": { "score": 55, @@ -127,7 +127,7 @@ } ], "methodology": "Code execution risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "network_access_control": { "score": 60, @@ -141,7 +141,7 @@ } ], "methodology": "Network boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 58, @@ -155,7 +155,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "malicious_site_protection": { "score": 68, @@ -169,7 +169,7 @@ } ], "methodology": "Malicious content protection testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data exposure analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_scraping_risk": { "score": 58, @@ -202,7 +202,7 @@ } ], "methodology": "Privacy risk assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "cookie_and_session_handling": { "score": 65, @@ -216,7 +216,7 @@ } ], "methodology": "Session security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "legal_compliance": { "score": 70, @@ -230,7 +230,7 @@ } ], "methodology": "Legal compliance assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "tracking_prevention": { "score": 75, @@ -244,7 +244,7 @@ } ], "methodology": "Privacy protection assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "action_visibility": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Action traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 90, @@ -291,7 +291,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "security_guidance": { "score": 65, @@ -305,7 +305,7 @@ } ], "methodology": "Security documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_efficiency": { "score": 65, @@ -338,7 +338,7 @@ } ], "methodology": "Resource utilization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "stability": { "score": 60, @@ -358,7 +358,7 @@ } ], "methodology": "Stability testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "browser_compatibility": { "score": 82, @@ -372,7 +372,7 @@ } ], "methodology": "Browser compatibility testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { "score": 58, @@ -389,10 +389,16 @@ "url": "https://github.com/modelcontextprotocol/servers-archived", "date": "2026-06-10", "value": "MCP server archived 2025-05-29; repository is read-only, issues and PRs are no longer accepted" + }, + { + "source": "Microsoft Playwright MCP releases", + "url": "https://github.com/microsoft/playwright-mcp/releases", + "date": "2026-07-09", + "value": "Recommended successor microsoft/playwright-mcp remains actively maintained with frequent releases (33k+ GitHub stars, releases through June 2026)" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } diff --git a/data/mcps/mcp-server-redis.json b/data/mcps/mcp-server-redis.json index 4ed337c..53e2d99 100644 --- a/data/mcps/mcp-server-redis.json +++ b/data/mcps/mcp-server-redis.json @@ -4,10 +4,10 @@ "name": "MCP Redis Server", "provider": "Anthropic (Archived)", "version": "2025.4.24", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for Redis cache and data structure operations. Enables AI models to interact with Redis for key-value operations, pub/sub messaging, lists, sets, sorted sets, and hashes. Essential for AI-powered caching strategies, session management, and real-time data workflows.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "ARCHIVED: Former Anthropic reference MCP server for Redis cache and data structure operations, archived 2025-05-29 to the servers-archived repository and no longer maintained (the archive README provides no security guarantees). Enabled AI models to interact with Redis for key-value operations, pub/sub messaging, lists, sets, sorted sets, and hashes. Redis now ships an official Redis MCP server (redis/mcp-redis, with redis/mcp-redis-cloud for Redis Cloud), which is the recommended replacement.", + "website": "https://github.com/modelcontextprotocol/servers-archived/tree/main/src/redis", "trust_vector": { "performance_reliability": { "overall_score": 88, @@ -24,7 +24,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "command_reliability": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Command execution testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_structure_accuracy": { "score": 92, @@ -52,7 +52,7 @@ } ], "methodology": "Data structure testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "connection_stability": { "score": 85, @@ -66,7 +66,7 @@ } ], "methodology": "Connection stability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pubsub_reliability": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Pub/sub testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "connection_string_exposure": { "score": 62, @@ -113,7 +113,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "command_injection_risk": { "score": 68, @@ -127,7 +127,7 @@ } ], "methodology": "Command injection testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_deletion_risk": { "score": 65, @@ -141,7 +141,7 @@ } ], "methodology": "Destructive operation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "acl_enforcement": { "score": 78, @@ -155,7 +155,7 @@ } ], "methodology": "ACL enforcement testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 72, @@ -169,7 +169,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "session_data_privacy": { "score": 60, @@ -202,7 +202,7 @@ } ], "methodology": "Session privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "key_pattern_exposure": { "score": 72, @@ -216,7 +216,7 @@ } ], "methodology": "Key privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "ttl_data_persistence": { "score": 70, @@ -244,7 +244,7 @@ } ], "methodology": "Data persistence assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -263,7 +263,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "command_visibility": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Command logging assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, @@ -291,7 +291,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "command_coverage_clarity": { "score": 68, @@ -305,12 +305,12 @@ } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 85, + "overall_score": 80, "criteria": { "ease_of_setup": { "score": 88, @@ -324,7 +324,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "command_performance": { "score": 95, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 85, @@ -352,7 +352,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "command_coverage": { "score": 78, @@ -366,21 +366,27 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { - "score": 75, - "confidence": "medium", + "score": 55, + "confidence": "high", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/modelcontextprotocol/servers/discussions", - "date": "2025-11-16", - "value": "Community-maintained with moderate activity" + "source": "MCP servers-archived repository", + "url": "https://github.com/modelcontextprotocol/servers-archived", + "date": "2026-07-09", + "value": "Reference Redis server archived 2025-05-29; repository read-only, no issues or PRs accepted, README states no security guarantees are provided for archived servers. Score lowered from 75 to align with other archived reference servers" + }, + { + "source": "Redis official MCP server", + "url": "https://github.com/redis/mcp-redis", + "date": "2026-07-09", + "value": "Redis maintains an official Redis MCP Server (redis/mcp-redis, PyPI package) as the actively maintained successor, covering hashes, lists, sets, sorted sets, streams, and vector search" } ], "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -399,7 +405,8 @@ "Redis connection strings with passwords accessible to AI", "Session data and user tokens commonly stored in Redis may be exposed", "Limited built-in audit logging capabilities", - "Key naming patterns may reveal application architecture" + "Key naming patterns may reveal application architecture", + "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; use the official Redis MCP server (redis/mcp-redis) instead" ], "metadata": { "license": "MIT", @@ -411,11 +418,12 @@ "Python" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", + "github_repo": "https://github.com/modelcontextprotocol/servers-archived", "api_dependency": "Redis client libraries (ioredis, redis-py)", "authentication": "Redis password, ACLs (Redis 6+)", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "None (Archived 2025-05-29)", + "replacement": "Official Redis MCP server (https://github.com/redis/mcp-redis); redis/mcp-redis-cloud for Redis Cloud" }, "use_case_ratings": { "code-generation": { @@ -468,6 +476,8 @@ "cache", "database", "mcp", - "model-context-protocol" + "model-context-protocol", + "archived", + "unmaintained" ] } diff --git a/data/mcps/mcp-server-s3.json b/data/mcps/mcp-server-s3.json index 5dd956f..553105b 100644 --- a/data/mcps/mcp-server-s3.json +++ b/data/mcps/mcp-server-s3.json @@ -4,10 +4,10 @@ "name": "MCP S3 Server", "provider": "Community", "version": "1.0.0", - "last_evaluated": "2025-11-09", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Community-maintained MCP server for AWS S3 storage operations. Enables AI models to upload, download, list, and manage objects in S3 buckets. Includes support for multipart uploads, object metadata, and bucket operations. Essential for AI-powered cloud storage management and data pipeline workflows.", - "website": "https://github.com/modelcontextprotocol/servers", + "description": "MCP servers for AWS S3 storage operations, enabling AI models to upload, download, list, and manage objects in S3 buckets. The landscape has shifted to official AWS options: the managed AWS MCP Server (GA 2026-05) covers all S3 APIs via its call_aws tool, and awslabs ships a dedicated S3 Tables MCP Server (read-only by default, write enabled only with --allow-write). Community S3 servers (e.g. aws-samples/sample-mcp-server-s3) remain available but are no longer the recommended path.", + "website": "https://github.com/awslabs/mcp/tree/main/src/s3-tables-mcp-server", "trust_vector": { "performance_reliability": { "overall_score": 85, @@ -24,7 +24,7 @@ } ], "methodology": "Upload success rate testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "download_performance": { "score": 87, @@ -38,7 +38,7 @@ } ], "methodology": "Download speed testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "large_file_handling": { "score": 82, @@ -52,7 +52,7 @@ } ], "methodology": "Large file transfer testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "listing_performance": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Listing performance testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 86, @@ -80,12 +80,12 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 71, + "overall_score": 73, "criteria": { "authentication_security": { "score": 80, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 63, @@ -113,7 +113,7 @@ } ], "methodology": "Credential security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "bucket_policy_enforcement": { "score": 78, @@ -127,21 +127,21 @@ } ], "methodology": "Policy enforcement testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "object_deletion_risk": { - "score": 62, + "score": 66, "confidence": "high", "evidence": [ { - "source": "Security Analysis", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "AI can delete objects and empty buckets if IAM permits" + "source": "AWS S3 Tables MCP Server", + "url": "https://awslabs.github.io/mcp/servers/s3-tables-mcp-server", + "date": "2026-07-09", + "value": "AI can delete objects and empty buckets if IAM permits; the official awslabs S3 Tables server mitigates this by defaulting to read-only and requiring an explicit --allow-write flag plus matching IAM permissions for writes" } ], - "methodology": "Destructive operation testing", - "last_verified": "2025-11-09" + "methodology": "Destructive operation testing; small raise from 62 reflecting the read-only-by-default posture of the current official server", + "last_verified": "2026-07-09" }, "public_access_risk": { "score": 68, @@ -155,7 +155,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 85, @@ -169,7 +169,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -188,7 +188,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "metadata_privacy": { "score": 68, @@ -202,7 +202,7 @@ } ], "methodology": "Metadata privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "bucket_structure_exposure": { "score": 72, @@ -216,7 +216,7 @@ } ], "methodology": "Structure privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -230,7 +230,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "encryption_support": { "score": 75, @@ -244,26 +244,26 @@ } ], "methodology": "Encryption support assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, "trust_transparency": { - "overall_score": 80, + "overall_score": 81, "criteria": { "documentation_quality": { - "score": 78, - "confidence": "medium", + "score": 82, + "confidence": "high", "evidence": [ { - "source": "S3 MCP Docs", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Good documentation but community-maintained with evolving coverage" + "source": "AWS S3 Tables MCP Server Documentation", + "url": "https://awslabs.github.io/mcp/servers/s3-tables-mcp-server", + "date": "2026-07-09", + "value": "Official awslabs documentation site covers setup, IAM requirements, and the --allow-write model; community S3 servers retain variable documentation" } ], - "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "methodology": "Documentation completeness review; raised from 78 as first-party AWS documentation now exists", + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 85, @@ -277,21 +277,21 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 88, "confidence": "high", "evidence": [ { - "source": "GitHub Repository", - "url": "https://github.com/modelcontextprotocol/servers", - "date": "2025-11-16", - "value": "Open source community implementation" + "source": "awslabs/mcp Repository", + "url": "https://github.com/awslabs/mcp/tree/main/src/s3-tables-mcp-server", + "date": "2026-07-09", + "value": "Official S3 Tables server is open source (Apache 2.0) in awslabs/mcp; community S3 implementations also open source" } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 70, @@ -305,7 +305,7 @@ } ], "methodology": "API documentation review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -324,7 +324,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "transfer_performance": { "score": 85, @@ -338,7 +338,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 88, @@ -352,7 +352,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_coverage": { "score": 80, @@ -366,21 +366,21 @@ } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_support": { - "score": 75, + "score": 80, "confidence": "medium", "evidence": [ { - "source": "GitHub Community", - "url": "https://github.com/modelcontextprotocol/servers/discussions", - "date": "2025-11-16", - "value": "Community-maintained with moderate activity" + "source": "awslabs/mcp Repository", + "url": "https://github.com/awslabs/mcp", + "date": "2026-07-09", + "value": "S3 Tables server is first-party maintained within the active awslabs/mcp suite; generic community S3 servers see moderate activity" } ], - "methodology": "Community support assessment", - "last_verified": "2025-11-09" + "methodology": "Community support assessment; raised from 75 reflecting first-party AWS maintenance of the recommended server", + "last_verified": "2026-07-09" } } } @@ -388,34 +388,38 @@ "strengths": [ "Built on highly reliable AWS S3 with 99.999999999% durability", "Comprehensive S3 operations (upload, download, list, delete, metadata)", + "Official first-party option now available: awslabs S3 Tables MCP Server, read-only by default", "Excellent for cloud storage automation and data pipelines", "Full operation auditability through CloudTrail", - "Open source community implementation", "Supports multipart uploads for large files up to 5TB" ], "limitations": [ "Downloaded object content and metadata exposed to LLM provider", - "AI can delete objects and modify ACLs within IAM permissions", - "AWS access keys and credentials accessible to AI", + "AI can delete objects and modify ACLs within IAM permissions (official S3 Tables server requires explicit --allow-write)", + "AWS access keys and credentials accessible to AI in locally configured community servers", "Object metadata and bucket structure may reveal sensitive information", "Can make objects public if IAM permissions allow", - "Encrypted data decrypted before transmission to LLM provider" + "Encrypted data decrypted before transmission to LLM provider", + "Generic community S3 servers are unofficial; AWS's recommended paths are the managed AWS MCP Server or the awslabs S3 Tables server" ], "metadata": { - "license": "MIT", + "license": "Apache 2.0 (awslabs S3 Tables server); community servers vary (MIT common)", "supported_platforms": [ - "All platforms with Node.js/Python" + "All platforms with Python (uvx awslabs.s3-tables-mcp-server) or Node.js (community servers)" ], "programming_languages": [ - "TypeScript", - "Python" + "Python", + "TypeScript" ], "mcp_version": "1.0", - "github_repo": "https://github.com/modelcontextprotocol/servers", - "api_dependency": "AWS SDK for JavaScript/Python", - "authentication": "AWS IAM credentials (Access Key ID, Secret Access Key)", + "github_repo": "https://github.com/awslabs/mcp/tree/main/src/s3-tables-mcp-server", + "alternative_repos": [ + "https://github.com/aws-samples/sample-mcp-server-s3" + ], + "api_dependency": "AWS SDK (boto3 / AWS SDK for JavaScript)", + "authentication": "AWS IAM credentials (access keys or IAM roles); write access gated behind --allow-write on the S3 Tables server", "first_release": "2024-11", - "maintained_by": "Community" + "maintained_by": "AWS (awslabs) and Community" }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-sentry.json b/data/mcps/mcp-server-sentry.json index 917a58c..734866c 100644 --- a/data/mcps/mcp-server-sentry.json +++ b/data/mcps/mcp-server-sentry.json @@ -3,11 +3,11 @@ "type": "mcp", "name": "MCP Sentry Server", "provider": "Sentry", - "version": "1.0.0", - "last_evaluated": "2025-11-08", + "version": "hosted remote + stdio", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official Sentry MCP server for error tracking and monitoring integration. Enables AI models to query errors, analyze stack traces, manage issues, track releases, and access performance data. Essential for AI-powered debugging, incident response, and application monitoring workflows.", - "website": "https://github.com/getsentry/sentry-mcp-server", + "description": "Official Sentry MCP server (getsentry/sentry-mcp) for error tracking and monitoring integration. Sentry hosts a managed remote at https://mcp.sentry.dev/mcp with OAuth authentication (nothing to install); a stdio mode with an auth token supports self-hosted Sentry. Enables AI models to query errors, analyze stack traces, manage issues, track releases, and access performance data. Also distributed as a Claude Code plugin.", + "website": "https://github.com/getsentry/sentry-mcp", "trust_vector": { "performance_reliability": { "overall_score": 87, @@ -24,7 +24,7 @@ } ], "methodology": "Query success rate testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "stack_trace_parsing_accuracy": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Parsing accuracy testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "real_time_monitoring": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "Real-time monitoring testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 85, @@ -74,13 +74,13 @@ "evidence": [ { "source": "Implementation Review", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Handles API errors with retry logic" } ], "methodology": "Error handling testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -96,10 +96,16 @@ "url": "https://docs.sentry.io/api/auth/", "date": "2025-11-16", "value": "Uses auth tokens or integration tokens with scoped permissions" + }, + { + "source": "Sentry MCP (official repo)", + "url": "https://github.com/getsentry/sentry-mcp", + "date": "2026-07-09", + "value": "Hosted remote at https://mcp.sentry.dev/mcp uses a browser OAuth flow with no locally stored token; stdio mode with a Sentry auth token supports self-hosted instances" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 65, @@ -113,7 +119,7 @@ } ], "methodology": "Token security analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "source_code_exposure_risk": { "score": 60, @@ -121,13 +127,13 @@ "evidence": [ { "source": "Security Analysis", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Stack traces may contain source code snippets and file paths" } ], "methodology": "Code exposure assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "issue_modification_control": { "score": 75, @@ -141,7 +147,7 @@ } ], "methodology": "Modification control testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "organization_access_control": { "score": 78, @@ -155,7 +161,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 85, @@ -169,7 +175,7 @@ } ], "methodology": "Audit logging review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -188,7 +194,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "user_data_in_errors": { "score": 58, @@ -196,13 +202,13 @@ "evidence": [ { "source": "Privacy Analysis", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Error context may include user IDs, emails, and session data" } ], "methodology": "PII exposure assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "source_code_privacy": { "score": 65, @@ -216,7 +222,7 @@ } ], "methodology": "Code privacy assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 68, @@ -230,7 +236,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "breadcrumb_data_privacy": { "score": 70, @@ -244,7 +250,7 @@ } ], "methodology": "Breadcrumb privacy assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -257,13 +263,13 @@ "evidence": [ { "source": "Sentry MCP Docs", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Comprehensive documentation from official Sentry team" } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 85, @@ -277,7 +283,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 90, @@ -285,13 +291,19 @@ "evidence": [ { "source": "GitHub Repository", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Open source implementation from Sentry" + }, + { + "source": "Sentry MCP (official repo)", + "url": "https://github.com/getsentry/sentry-mcp", + "date": "2026-07-09", + "value": "Source available under FSL-1.1-Apache-2.0 (Functional Source License, converts to Apache-2.0 after two years), not MIT" } ], "methodology": "Source code review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 73, @@ -299,13 +311,13 @@ "evidence": [ { "source": "MCP Server Documentation", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Clear documentation of supported Sentry operations" } ], "methodology": "API documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } }, @@ -318,13 +330,19 @@ "evidence": [ { "source": "Setup Documentation", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Simple setup requiring Sentry auth token" + }, + { + "source": "Sentry MCP hosted service", + "url": "https://mcp.sentry.dev/", + "date": "2026-07-09", + "value": "Hosted remote requires no install: add https://mcp.sentry.dev/mcp and complete OAuth; also installable as a Claude Code plugin (sentry-mcp)" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_performance": { "score": 85, @@ -338,7 +356,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "reliability": { "score": 90, @@ -352,7 +370,7 @@ } ], "methodology": "Reliability analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "operation_coverage": { "score": 82, @@ -360,13 +378,13 @@ "evidence": [ { "source": "Sentry MCP Server", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Covers errors, issues, releases, performance, and projects" } ], "methodology": "Feature coverage assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "official_support": { "score": 88, @@ -374,13 +392,13 @@ "evidence": [ { "source": "Sentry Team", - "url": "https://github.com/getsentry/sentry-mcp-server", + "url": "https://github.com/getsentry/sentry-mcp", "date": "2025-11-16", "value": "Officially maintained by Sentry with active support" } ], "methodology": "Maintainer support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } } } @@ -388,6 +406,7 @@ "strengths": [ "Comprehensive error tracking and monitoring capabilities", "Official Sentry implementation with active support", + "Managed hosted remote (https://mcp.sentry.dev/mcp) with OAuth; no token storage or install required", "Accurate stack trace parsing and symbolication", "Real-time error notifications via webhooks", "Excellent for AI-powered debugging and incident response", @@ -402,20 +421,31 @@ "Requires careful data scrubbing configuration to avoid PII exposure" ], "metadata": { - "license": "MIT", + "license": "FSL-1.1-Apache-2.0", "supported_platforms": [ - "All platforms with Node.js/Python" + "Hosted remote (https://mcp.sentry.dev/mcp)", + "All platforms with Node.js (stdio)" ], "programming_languages": [ - "TypeScript", - "Python" + "TypeScript" ], "mcp_version": "1.0", - "github_repo": "https://github.com/getsentry/sentry-mcp-server", + "github_repo": "https://github.com/getsentry/sentry-mcp", "api_dependency": "Sentry REST API", - "authentication": "Sentry Auth Token or Integration Token", + "authentication": "OAuth (hosted remote) or Sentry Auth Token (stdio / self-hosted Sentry)", + "remote_endpoint": "https://mcp.sentry.dev/mcp", "first_release": "2024-11", - "maintained_by": "Sentry" + "maintained_by": "Sentry", + "status": "Active - official Sentry server with managed hosted remote", + "transport_types": [ + "streamable-http (hosted remote)", + "stdio" + ], + "installation_methods": [ + "Hosted remote (no install)", + "npm (stdio)", + "Claude Code plugin (sentry-mcp)" + ] }, "use_case_ratings": { "code-generation": { @@ -468,6 +498,8 @@ "error-tracking", "monitoring", "mcp", - "model-context-protocol" + "model-context-protocol", + "official", + "remote-server" ] } diff --git a/data/mcps/mcp-server-sequential-thinking.json b/data/mcps/mcp-server-sequential-thinking.json index d6f6335..ae5fb66 100644 --- a/data/mcps/mcp-server-sequential-thinking.json +++ b/data/mcps/mcp-server-sequential-thinking.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Sequential Thinking Server", "provider": "Anthropic", - "version": "2025.7.1", - "last_evaluated": "2026-06-10", + "version": "2026.7.4", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official MCP reference server enabling dynamic, extended reasoning and problem-solving sequences. Allows AI models to create structured thinking processes, break down complex problems, and maintain context across multi-step reasoning chains. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", + "description": "Official MCP reference server enabling dynamic, extended reasoning and problem-solving sequences: structured thinking processes, decomposition of complex problems, and context across multi-step reasoning chains. One of the seven reference servers still maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09 (latest stable spec 2025-11-25; 2026-07-28 revision at release-candidate stage).", "website": "https://modelcontextprotocol.io/docs/servers/sequential-thinking", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Reasoning quality assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "context_preservation": { "score": 83, @@ -38,7 +38,7 @@ } ], "methodology": "Context retention testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "step_execution_reliability": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Execution reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "complex_problem_handling": { "score": 78, @@ -66,7 +66,7 @@ } ], "methodology": "Problem-solving capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Isolation boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_exposure_risk": { "score": 85, @@ -113,7 +113,7 @@ } ], "methodology": "Data flow security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "prompt_injection_resistance": { "score": 82, @@ -127,7 +127,7 @@ } ], "methodology": "Injection attack testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reasoning_manipulation_risk": { "score": 88, @@ -141,7 +141,7 @@ } ], "methodology": "Manipulation vulnerability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_consumption_limits": { "score": 90, @@ -155,7 +155,7 @@ } ], "methodology": "Resource limit testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data privacy analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "thought_process_exposure": { "score": 78, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "context_data_retention": { "score": 80, @@ -202,7 +202,7 @@ } ], "methodology": "Data retention assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 80, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reasoning_visibility": { "score": 90, @@ -249,7 +249,7 @@ } ], "methodology": "Process visibility assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -263,7 +263,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "experimental_status_disclosure": { "score": 70, @@ -277,7 +277,7 @@ } ], "methodology": "Status disclosure review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reasoning_performance": { "score": 72, @@ -310,7 +310,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 75, @@ -324,7 +324,7 @@ } ], "methodology": "Stability analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_maturity": { "score": 70, @@ -338,7 +338,7 @@ } ], "methodology": "Maturity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 68, @@ -355,10 +355,16 @@ "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", "date": "2025-12-09", "value": "sequential-thinking is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" + }, + { + "source": "npm - @modelcontextprotocol/server-sequential-thinking", + "url": "https://www.npmjs.com/package/@modelcontextprotocol/server-sequential-thinking", + "date": "2026-07-09", + "value": "Latest release 2026.7.4 (published 2026-07-04); still listed among the seven maintained reference servers in modelcontextprotocol/servers README (repo ~88,300 stars, not archived)" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -378,7 +384,7 @@ "Reasoning steps exposed to LLM provider", "Relatively low community adoption due to specialized use case", "Documentation and best practices still evolving", - "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" + "STATUS 2026-07-09: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); latest npm release 2026.7.4 (2026-07-04); no known CVEs against this server; governance under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -391,7 +397,7 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/modelcontextprotocol/servers", - "github_stars": 58700, + "github_stars": 88259, "api_dependency": "None (MCP protocol only)", "authentication": "None required", "first_release": "2024-11", diff --git a/data/mcps/mcp-server-serena.json b/data/mcps/mcp-server-serena.json index ac841bd..aec3e69 100644 --- a/data/mcps/mcp-server-serena.json +++ b/data/mcps/mcp-server-serena.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "Serena MCP", "provider": "Oraios AI", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "1.5.3", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Open-source semantic coding toolkit from Oraios AI that turns any MCP-capable agent into an IDE-grade coding assistant. Uses language servers (LSP) for symbol-level code navigation and editing — find_symbol, find_referencing_symbols, precise symbol edits — plus project memory and shell execution. High-privilege local tooling: full filesystem and shell access.", + "description": "Open-source semantic coding toolkit from Oraios AI that turns any MCP-capable agent into an IDE-grade coding assistant. Uses language servers (LSP) for symbol-level code navigation and editing — find_symbol, find_referencing_symbols, precise symbol edits, plus refactoring tools (rename/move/inline) — with project memory and shell execution across 40+ languages. An optional paid JetBrains-plugin backend adds interactive debugging. High-privilege local tooling: full filesystem and shell access.", "website": "https://github.com/oraios/serena", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Accuracy testing of symbol resolution and reference finding against IDE ground truth in multi-file projects", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 84, @@ -38,7 +38,7 @@ } ], "methodology": "Hands-on testing of symbolic read/edit operations across project states", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "language_server_stability": { "score": 76, @@ -52,7 +52,7 @@ } ], "methodology": "Review of reported language-server issues and stress testing on large repositories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Error-path testing including LSP crashes, unindexed files, and invalid edits", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "large_project_handling": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Token-efficiency and navigation testing on repositories with 100k+ lines", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Capability analysis of shell execution tooling and its abuse potential under prompt injection", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "filesystem_access_risk": { "score": 50, @@ -113,7 +113,7 @@ } ], "methodology": "Analysis of file read/write tool boundaries and project-scoping enforcement", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sandboxing_isolation": { "score": 45, @@ -127,7 +127,7 @@ } ], "methodology": "Review of process isolation, privilege boundaries, and available containment options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_exposure_risk": { "score": 58, @@ -141,7 +141,7 @@ } ], "methodology": "Analysis of secret-reachability via file and shell tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 52, @@ -155,7 +155,7 @@ } ], "methodology": "Authorization boundary analysis of write and execution tools, including available mode-based restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of tool outputs to the LLM provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 60, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy controls assessment of file-content handling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_data_control": { "score": 88, @@ -202,7 +202,7 @@ } ], "methodology": "Review of local execution model, memory storage, and data residency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 85, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing pathway analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -249,7 +249,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -263,7 +263,7 @@ } ], "methodology": "Logging and observability assessment including the built-in dashboard", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "project_memory_transparency": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Review of memory persistence format, location, and influence on agent sessions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment including language-server prerequisites", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "performance": { "score": 80, @@ -310,7 +310,7 @@ } ], "methodology": "Latency and token-efficiency benchmarking on indexed projects", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 88, @@ -319,12 +319,12 @@ { "source": "Serena tool suite", "url": "https://github.com/oraios/serena", - "date": "2026-06-10", - "value": "Comprehensive coding toolkit: symbol search/references, symbol-level editing, pattern search, file operations, shell execution, project memory, and onboarding — spanning 20+ languages via LSP" + "date": "2026-07-09", + "value": "Comprehensive coding toolkit: symbol search/references, symbol-level editing, refactoring (rename/move/inline), pattern search, file operations, shell execution, project memory, and onboarding — spanning 40+ languages via LSP, with an optional paid JetBrains-plugin backend adding interactive debugging" } ], "methodology": "Feature completeness assessment against IDE-grade coding-agent needs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 88, @@ -333,12 +333,12 @@ { "source": "GitHub API", "url": "https://api.github.com/repos/oraios/serena", - "date": "2026-06-10", - "value": "25,204 stars as of 2026-06-10; widely adopted as a free, open-source way to add semantic code tools to Claude, and other MCP-capable agents" + "date": "2026-07-09", + "value": "~26,300 stars as of 2026-07-09; widely adopted as a free, open-source way to add semantic code tools to Claude, and other MCP-capable agents" } ], "methodology": "Adoption metrics and community-activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "maintenance_activity": { "score": 90, @@ -347,12 +347,12 @@ { "source": "GitHub repository activity", "url": "https://github.com/oraios/serena/commits/main", - "date": "2026-06-10", - "value": "Very active development by Oraios AI with frequent releases, expanding language support, and responsive issue triage" + "date": "2026-07-09", + "value": "Very active development by Oraios AI with frequent releases (latest v1.5.3, 2026-05-26; 3,000+ commits), expanding language support, and responsive issue triage" } ], "methodology": "Commit frequency and release-cadence analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -362,7 +362,8 @@ "Symbol-level reading/editing slashes token usage on large codebases versus whole-file approaches", "Fully open source (MIT), fully local — no hosted backend, no telemetry, no API costs", "Project memory system persists codebase knowledge across sessions as inspectable markdown", - "Broad language coverage (20+ languages) and active maintenance (25.2k stars)", + "Broad language coverage (40+ languages) and active maintenance (~26.3k stars)", + "Refactoring tools (rename/move/inline) and optional JetBrains-plugin backend with interactive debugging", "Modes/contexts allow restricting the tool surface, including disabling shell execution" ], "limitations": [ @@ -383,7 +384,8 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/oraios/serena", - "github_stars": 25204, + "github_stars": 26300, + "latest_release": "v1.5.3 (2026-05-26)", "package": "serena-agent (PyPI)", "api_dependency": "Local language servers (LSP) per language", "authentication": "None required (local operation)", diff --git a/data/mcps/mcp-server-shadcn.json b/data/mcps/mcp-server-shadcn.json index 0d76b22..4af9f5a 100644 --- a/data/mcps/mcp-server-shadcn.json +++ b/data/mcps/mcp-server-shadcn.json @@ -3,11 +3,11 @@ "type": "mcp", "name": "shadcn MCP Server", "provider": "shadcn", - "version": "CLI 3.x", - "last_evaluated": "2026-06-10", + "version": "CLI 4.x", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "MCP server built into the shadcn CLI (run via npx shadcn@latest mcp) that lets AI agents browse, search, and install UI components from any shadcn-compatible registry, including private registries, using @registry/name namespacing. Shipped with CLI 3.0 in August 2025.", - "website": "https://ui.shadcn.com/docs/registry/mcp", + "description": "MCP server built into the shadcn CLI (run via npx shadcn@latest mcp) that lets AI agents browse, search, and install UI components from any shadcn-compatible registry, including private registries, using @registry/name namespacing. Shipped with CLI 3.0 in August 2025; CLI v4 (March 2026) added shadcn/skills for coding agents, design-system presets, and --dry-run/--diff/--view flags to inspect registry changes before installation.", + "website": "https://ui.shadcn.com/docs/mcp", "trust_vector": { "performance_reliability": { "overall_score": 85, @@ -18,13 +18,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Namespaced @registry/name resolution reliably targets components across configured registries" } ], "methodology": "Registry resolution testing across namespaces", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "installation_success_rate": { "score": 87, @@ -38,7 +38,7 @@ } ], "methodology": "Component installation success testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "component_search_quality": { "score": 84, @@ -46,13 +46,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Browse and search tools surface components with metadata from any shadcn-compatible registry" } ], "methodology": "Search relevance assessment over public registries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Failure mode and error message testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -85,7 +85,7 @@ } ], "methodology": "Supply chain threat modeling of registry-sourced code installation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "code_installation_control": { "score": 62, @@ -93,13 +93,19 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Agent-initiated installs write component files and add npm dependencies; review depends on the MCP client's approval flow and code review practices" + }, + { + "source": "shadcn CLI v4 Changelog", + "url": "https://ui.shadcn.com/docs/changelog/2026-03-cli-v4", + "date": "2026-07-09", + "value": "CLI v4 (March 2026) adds --dry-run, --diff, and --view flags so registry changes can be inspected before installation, improving pre-install review of agent-driven installs" } ], "methodology": "Write-action control and approval flow analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "registry_authentication": { "score": 75, @@ -113,7 +119,7 @@ } ], "methodology": "Registry authentication mechanism review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_handling": { "score": 78, @@ -127,7 +133,7 @@ } ], "methodology": "Credential storage and flow analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 60, @@ -141,7 +147,7 @@ } ], "methodology": "Blast radius assessment of install actions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -154,13 +160,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Server only fetches public or configured registry metadata and component source; it does not read or transmit project code externally" } ], "methodology": "Data flow analysis of registry requests", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 80, @@ -174,7 +180,7 @@ } ], "methodology": "Sensitive data surface assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 78, @@ -188,7 +194,7 @@ } ], "methodology": "Outbound request and data sharing review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "local_execution_privacy": { "score": 90, @@ -196,13 +202,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Runs locally over stdio via npx shadcn@latest mcp with no hosted service or telemetry pipeline collecting project data" } ], "methodology": "Local execution and telemetry review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -215,13 +221,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Clear official docs covering MCP setup per client, registry configuration, and namespacing, plus a detailed CLI 3.0 changelog" } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 80, @@ -235,7 +241,7 @@ } ], "methodology": "Traceability of agent actions assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -249,7 +255,7 @@ } ], "methodology": "Source code and license review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_coverage_clarity": { "score": 85, @@ -257,13 +263,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Tool surface is small and well-defined: browse, search, and install components from configured registries" } ], "methodology": "Tool surface documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -276,13 +282,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "One-line stdio config (npx shadcn@latest mcp) with documented one-command setup for major MCP clients" } ], "methodology": "Setup complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "performance": { "score": 85, @@ -296,21 +302,21 @@ } ], "methodology": "Operation latency characterization", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 86, "confidence": "medium", "evidence": [ { - "source": "shadcn CLI 3.0 Changelog", - "url": "https://ui.shadcn.com/docs/changelog/2025-08-cli-3-mcp", - "date": "2026-06-10", - "value": "Built on the actively maintained shadcn CLI with frequent releases since the August 2025 3.0 launch" + "source": "shadcn CLI v4 Changelog", + "url": "https://ui.shadcn.com/docs/changelog/2026-03-cli-v4", + "date": "2026-07-09", + "value": "Built on the actively maintained shadcn CLI with frequent releases since the August 2025 3.0 launch; CLI v4 shipped March 2026 (npm shadcn at v4.13.0 as of 2026-07-09) with skills, presets, and expanded init templates" } ], "methodology": "Maintenance cadence and stability analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 84, @@ -318,13 +324,13 @@ "evidence": [ { "source": "shadcn Registry MCP Documentation", - "url": "https://ui.shadcn.com/docs/registry/mcp", + "url": "https://ui.shadcn.com/docs/mcp", "date": "2026-06-10", "value": "Covers discovery and installation across public and private shadcn-compatible registries; scoped to UI components rather than general tooling" } ], "methodology": "Feature completeness assessment within its domain", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 93, @@ -338,7 +344,7 @@ } ], "methodology": "Ecosystem adoption analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -368,6 +374,7 @@ ], "github_repo": "https://github.com/shadcn-ui/ui", "package_name": "shadcn", + "package_version": "4.13.0", "installation": "npx shadcn@latest mcp", "authentication": "Optional registry tokens for private registries", "first_release": "2025-08", diff --git a/data/mcps/mcp-server-slack.json b/data/mcps/mcp-server-slack.json index 3570609..0b06e7a 100644 --- a/data/mcps/mcp-server-slack.json +++ b/data/mcps/mcp-server-slack.json @@ -4,9 +4,9 @@ "name": "MCP Slack Server", "provider": "Anthropic (Archived)", "version": "2025.4.24", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "ARCHIVED: Former Anthropic reference MCP server for Slack workspace interaction, archived 2025-05-29 to the servers-archived repository. NO LONGER MAINTAINED and no security guarantees are provided for archived servers. Its replacement in the MCP ecosystem is a third-party maintained Slack MCP server; evaluate maintained alternatives before deploying.", + "description": "ARCHIVED: Former Anthropic reference MCP server for Slack workspace interaction, archived 2025-05-29 to the servers-archived repository. NO LONGER MAINTAINED and no security guarantees are provided for archived servers. Slack now ships an official Slack MCP server, generally available since 2026-02-17 (announced at Dreamforce October 2025) with expanded tools added 2026-05-13; it is the recommended replacement.", "website": "https://api.slack.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Message delivery success testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Search quality testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 78, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "real_time_updates": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Event delivery latency testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -80,7 +80,7 @@ } ], "methodology": "Error handling testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Authentication security review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "message_access_risk": { "score": 68, @@ -113,7 +113,7 @@ } ], "methodology": "Access boundary testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "posting_control": { "score": 70, @@ -127,7 +127,7 @@ } ], "methodology": "Message posting capability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_exfiltration_risk": { "score": 72, @@ -141,7 +141,7 @@ } ], "methodology": "Data exposure analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 82, @@ -155,7 +155,7 @@ } ], "methodology": "Audit capabilities review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data exposure analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 62, @@ -188,7 +188,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "workspace_admin_control": { "score": 78, @@ -202,7 +202,7 @@ } ], "methodology": "Enterprise control assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_retention": { "score": 72, @@ -216,7 +216,7 @@ } ], "methodology": "Data retention policy review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "compliance_readiness": { "score": 73, @@ -230,7 +230,7 @@ } ], "methodology": "Compliance framework review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -249,7 +249,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -263,7 +263,7 @@ } ], "methodology": "Operation traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "permission_transparency": { "score": 80, @@ -277,7 +277,7 @@ } ], "methodology": "Permission clarity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "mcp_implementation": { "score": 55, @@ -288,10 +288,16 @@ "url": "https://github.com/modelcontextprotocol/servers-archived", "date": "2026-06-10", "value": "Reference Slack server archived 2025-05-29; repository read-only, README states no security guarantees are provided for archived servers" + }, + { + "source": "Slack Developer Docs changelog", + "url": "https://docs.slack.dev/changelog/2026/02/17/slack-mcp/", + "date": "2026-07-09", + "value": "Slack's official MCP server reached general availability 2026-02-17, and additional tools (reactions, channel creation, member listing, file reading) shipped 2026-05-13; recommended replacement for the archived reference server" } ], "methodology": "Implementation documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -310,7 +316,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "api_reliability": { "score": 88, @@ -324,7 +330,7 @@ } ], "methodology": "Uptime analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "response_time": { "score": 82, @@ -338,7 +344,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 55, @@ -358,7 +364,7 @@ } ], "methodology": "Feature completeness assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "cost": { "score": 75, @@ -372,7 +378,7 @@ } ], "methodology": "Cost analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } } @@ -390,7 +396,7 @@ "No built-in content filtering or PII protection", "Risk of exposing confidential team communications", "Rate limits can be restrictive for high-volume operations", - "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; replacement is third-party maintained", + "ARCHIVED 2025-05-29: reference server unmaintained with no security guarantees; use Slack's official MCP server (GA February 2026) instead", "Compliance challenges when sharing workspace data with third parties" ], "metadata": { @@ -410,7 +416,8 @@ "rate_limits": "Tier-based (1+ requests per second)", "first_release": "2024-11", "maintained_by": "None (Archived 2025-05-29)", - "repository": "https://github.com/modelcontextprotocol/servers-archived" + "repository": "https://github.com/modelcontextprotocol/servers-archived", + "replacement": "Official Slack MCP server (GA 2026-02-17, https://docs.slack.dev/changelog/2026/02/17/slack-mcp/)" }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-sqlite.json b/data/mcps/mcp-server-sqlite.json index 4fe5037..b681c31 100644 --- a/data/mcps/mcp-server-sqlite.json +++ b/data/mcps/mcp-server-sqlite.json @@ -4,7 +4,7 @@ "name": "MCP SQLite Server", "provider": "Anthropic (Archived)", "version": "2025.4.24", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "ARCHIVED WITH UNPATCHED VULNERABILITY: Former Anthropic reference MCP server for SQLite, archived 2025-05-29 to the servers-archived repository. The archived code contains the same class of unpatched SQL injection flaw disclosed by Trend Micro (June 2025) and will not receive fixes; the archive README provides no security guarantees. Not recommended for new deployments.", "website": "https://modelcontextprotocol.io/docs/servers/sqlite", @@ -24,7 +24,7 @@ } ], "methodology": "Query performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "schema_introspection": { "score": 92, @@ -38,7 +38,7 @@ } ], "methodology": "Schema discovery testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_integrity": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "Transaction testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "concurrent_access": { "score": 82, @@ -66,7 +66,7 @@ } ], "methodology": "Concurrency testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "error_handling": { "score": 88, @@ -80,7 +80,7 @@ } ], "methodology": "Error scenario testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Vulnerability disclosure review and code analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "file_access_control": { "score": 70, @@ -113,7 +113,7 @@ } ], "methodology": "Access control testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_modification_risk": { "score": 50, @@ -133,7 +133,7 @@ } ], "methodology": "Write operation risk assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "encryption_support": { "score": 65, @@ -147,7 +147,7 @@ } ], "methodology": "Encryption capabilities review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "audit_logging": { "score": 75, @@ -161,7 +161,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -180,7 +180,7 @@ } ], "methodology": "Data flow and exposure analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "pii_protection": { "score": 62, @@ -194,7 +194,7 @@ } ], "methodology": "Privacy controls assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "local_processing": { "score": 88, @@ -208,7 +208,7 @@ } ], "methodology": "Data processing architecture review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_retention": { "score": 75, @@ -222,7 +222,7 @@ } ], "methodology": "Data retention control review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -241,7 +241,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "query_visibility": { "score": 92, @@ -255,7 +255,7 @@ } ], "methodology": "Query traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_code": { "score": 95, @@ -269,7 +269,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "sqlite_transparency": { "score": 88, @@ -283,7 +283,7 @@ } ], "methodology": "Database engine transparency review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -302,7 +302,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "performance": { "score": 92, @@ -316,7 +316,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_efficiency": { "score": 94, @@ -330,7 +330,7 @@ } ], "methodology": "Resource utilization testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "portability": { "score": 90, @@ -344,7 +344,7 @@ } ], "methodology": "Cross-platform testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "maintenance": { "score": 58, @@ -359,12 +359,12 @@ { "source": "MCP servers-archived repository", "url": "https://github.com/modelcontextprotocol/servers-archived", - "date": "2026-06-10", - "value": "MCP server archived 2025-05-29; repository read-only, no maintainer, no bug fixes or security patches will be released" + "date": "2026-07-09", + "value": "Re-verified 2026-07-09: repository archived 2025-05-29 and read-only; README states 'NO SECURITY GUARANTEES ARE PROVIDED FOR THESE ARCHIVED SERVERS' and no security updates or bug fixes will be provided; sqlite remains in the archived list and is absent from the seven maintained servers in modelcontextprotocol/servers" } ], "methodology": "Maintenance overhead assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } diff --git a/data/mcps/mcp-server-stripe.json b/data/mcps/mcp-server-stripe.json index db40abc..d757407 100644 --- a/data/mcps/mcp-server-stripe.json +++ b/data/mcps/mcp-server-stripe.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "Stripe MCP Server", "provider": "Stripe", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Stripe's official MCP server from the open-source agent toolkit. Exposes tools for customers, products, prices, payment links, invoices, refunds, balance, disputes, and subscriptions plus Stripe documentation search. Available as the @stripe/mcp stdio package or the hosted remote server at mcp.stripe.com with OAuth.", + "description": "Stripe's official MCP server (repo renamed from stripe/agent-toolkit to stripe/ai). The hosted server at mcp.stripe.com (OAuth or restricted-key bearer token) exposes generic API search/read/write tools plus resource search, docs search, an implementation planner, refunds, and account info, spanning payments, subscriptions, invoices, products, disputes, and balance. Treasury/payout tools are in preview by request. Also available as the @stripe/mcp stdio package (v0.3.3).", "website": "https://docs.stripe.com/mcp", "trust_vector": { "performance_reliability": { @@ -24,21 +24,21 @@ } ], "methodology": "API stability and uptime analysis of the underlying Stripe API", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 88, "confidence": "high", "evidence": [ { - "source": "Stripe Agent Toolkit Repository", - "url": "https://github.com/stripe/agent-toolkit", + "source": "Stripe AI Repository (formerly agent-toolkit)", + "url": "https://github.com/stripe/ai", "date": "2026-06-10", "value": "Tools are thin, well-tested wrappers over stable Stripe API endpoints (customers, invoices, payment links, refunds, subscriptions)" } ], "methodology": "Operation success testing across the documented tool set in test mode", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 84, @@ -52,7 +52,7 @@ } ], "methodology": "Rate limiting behavior testing under sustained tool-call load", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 84, @@ -66,21 +66,21 @@ } ], "methodology": "Relevance assessment of documentation search results for common integration queries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 84, "confidence": "medium", "evidence": [ { - "source": "Stripe Agent Toolkit Repository", - "url": "https://github.com/stripe/agent-toolkit", + "source": "Stripe AI Repository (formerly agent-toolkit)", + "url": "https://github.com/stripe/ai", "date": "2026-06-10", "value": "Surfaces Stripe's structured error objects (decline codes, validation errors) to the agent, enabling informed retries" } ], "methodology": "Error handling testing with invalid parameters, missing permissions, and declined operations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -94,12 +94,12 @@ { "source": "Stripe MCP Documentation", "url": "https://docs.stripe.com/mcp", - "date": "2026-06-10", - "value": "Local server authenticates with Stripe API keys (restricted keys supported); hosted server at https://mcp.stripe.com uses OAuth with explicit consent" + "date": "2026-07-09", + "value": "Local server authenticates with Stripe API keys (restricted keys supported); hosted server at https://mcp.stripe.com uses OAuth (recommended, with sessions viewable/revocable in Dashboard) or bearer tokens with restricted keys; sandbox and live MCP access are separately manageable by admins" } ], "methodology": "Authentication mechanism review for stdio and hosted transports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scope_limitation": { "score": 88, @@ -113,7 +113,7 @@ } ], "methodology": "Permission scope testing with restricted keys across read and write tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 75, @@ -127,7 +127,7 @@ } ], "methodology": "Token storage and exposure-surface analysis for local configuration files", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "action_auditability": { "score": 88, @@ -141,7 +141,7 @@ } ], "methodology": "Audit logging review of API request logs and event history", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { "score": 60, @@ -150,12 +150,12 @@ { "source": "Stripe MCP Documentation", "url": "https://docs.stripe.com/mcp", - "date": "2026-06-10", - "value": "Tools include money-movement operations (refunds, payment links, invoices, subscription changes); an agent acting on injected or mistaken instructions can cause direct financial impact without human confirmation" + "date": "2026-07-09", + "value": "Tools include money-movement operations (refunds, generic stripe_api_write across invoices, subscriptions, payment links); Stripe's docs now explicitly recommend enabling client-side human confirmation of tools to mitigate prompt injection, but there is still no server-side confirmation step" } ], "methodology": "Threat modeling of write-capable financial tools; strongest case among evaluated servers for read-only defaults and human-in-the-loop confirmation on writes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of tool results containing customer and payment data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 80, @@ -188,7 +188,7 @@ } ], "methodology": "Review of Stripe compliance posture and the data classes reachable via the tool surface", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "organization_data_control": { "score": 82, @@ -202,7 +202,7 @@ } ], "methodology": "Access control review of key management and OAuth grant administration", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 74, @@ -216,7 +216,7 @@ } ], "methodology": "Analysis of downstream data sharing once tool results leave Stripe", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 88, @@ -249,21 +249,21 @@ } ], "methodology": "Logging and traceability assessment via Dashboard logs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 92, "confidence": "high", "evidence": [ { - "source": "Stripe Agent Toolkit Repository", - "url": "https://github.com/stripe/agent-toolkit", - "date": "2026-06-10", - "value": "MIT-licensed open source (approximately 1,601 stars); tool implementations are fully auditable in the stripe/agent-toolkit repository (being renamed stripe/ai)" + "source": "Stripe AI Repository (formerly agent-toolkit)", + "url": "https://github.com/stripe/ai", + "date": "2026-07-09", + "value": "MIT-licensed open source (1,650 stars); repository rename to stripe/ai is complete and tool implementations remain fully auditable, alongside new @stripe/ai-sdk and @stripe/token-meter packages" } ], "methodology": "Source code review of the published toolkit and MCP package", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 85, @@ -272,12 +272,12 @@ { "source": "Stripe MCP Documentation", "url": "https://docs.stripe.com/mcp", - "date": "2026-06-10", - "value": "Tool list is explicitly enumerated (customers, products, prices, payment links, invoices, refunds, balance, disputes, subscriptions, doc search) with per-tool enablement flags" + "date": "2026-07-09", + "value": "Tool list restructured and explicitly enumerated: stripe_api_search/details/read/write meta-tools, search/fetch resources, documentation search, implementation planner, create_refund, and account info, with supported API methods listed (customers, charges, PaymentIntents, Checkout Sessions, invoices, subscriptions, coupons, promotion codes, products, prices, payment links, disputes, portal configurations, balance)" } ], "methodology": "Comparison of documented tool surface against the shipped package", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment for both stdio and hosted transports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 86, @@ -310,7 +310,7 @@ } ], "methodology": "Latency observation across representative tool calls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 90, @@ -324,35 +324,35 @@ } ], "methodology": "Uptime analysis of Stripe infrastructure", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 84, "confidence": "high", "evidence": [ { - "source": "Stripe Agent Toolkit Repository", - "url": "https://github.com/stripe/agent-toolkit", - "date": "2026-06-10", - "value": "Covers core billing and payments objects plus doc search; advanced surfaces (Connect, Treasury, Radar) are not fully exposed as tools" + "source": "Stripe MCP Documentation", + "url": "https://docs.stripe.com/mcp", + "date": "2026-07-09", + "value": "Coverage expanded to charges, PaymentIntents, Checkout Sessions, coupons, promotion codes, and portal configurations in addition to core billing objects; Treasury/payout tools are in preview by request (mcp@stripe.com), while Connect and Radar remain unexposed" } ], "methodology": "Feature completeness assessment against the full Stripe API surface", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 82, "confidence": "medium", "evidence": [ { - "source": "Stripe Agent Toolkit Repository", - "url": "https://github.com/stripe/agent-toolkit", - "date": "2026-06-10", - "value": "Approximately 1,601 GitHub stars with active first-party maintenance; widely referenced as the canonical payments MCP server" + "source": "Stripe AI Repository (formerly agent-toolkit)", + "url": "https://github.com/stripe/ai", + "date": "2026-07-09", + "value": "1,650 GitHub stars with active first-party maintenance (last push July 2026); widely referenced as the canonical payments MCP server" } ], "methodology": "Community activity and adoption analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -371,15 +371,17 @@ "No built-in human confirmation step on write operations; must be enforced by the client", "Customer PII in tool results is shared with the LLM provider", "Unrestricted secret keys in stdio config are a severe single point of failure", - "Tool coverage omits advanced surfaces like Connect and Treasury", + "Tool coverage omits Connect and Radar; Treasury/payout tools are preview-only by request", "Agent errors in live mode can require manual financial remediation" ], "metadata": { - "repository": "https://github.com/stripe/agent-toolkit", + "repository": "https://github.com/stripe/ai", + "previous_repository": "https://github.com/stripe/agent-toolkit", "package_name": "@stripe/mcp", + "package_version": "0.3.3", "license": "MIT", "maintained_by": "Stripe", - "github_stars": 1601, + "github_stars": 1650, "remote_endpoint": "https://mcp.stripe.com", "authentication": "Stripe API keys / Restricted API Keys (stdio); OAuth (hosted)", "transport_types": [ diff --git a/data/mcps/mcp-server-supabase.json b/data/mcps/mcp-server-supabase.json index e042e6c..2b04ef4 100644 --- a/data/mcps/mcp-server-supabase.json +++ b/data/mcps/mcp-server-supabase.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Supabase Server", "provider": "Supabase", - "version": "2025.1.0", - "last_evaluated": "2025-01-14", + "version": "0.x (supabase/mcp monorepo, pre-1.0)", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "MCP server enabling AI models to interact with Supabase backend services. Provides schema design, database migrations, SQL query execution, TypeScript type generation, and real-time subscription management for building full-stack applications.", + "description": "Official Supabase MCP server: SQL execution, schema/migrations, type generation, Edge Function deploys, logs, storage, branching. Primarily a hosted remote at https://mcp.supabase.com/mcp with OAuth 2.1 dynamic client registration (PAT only for CI/CD), plus stdio and self-hosted modes. After 2025 prompt-injection/data-leak research, Supabase added mitigations: read-only mode, project_ref scoping, feature-group tool restrictions, and wrapping SQL results to deter embedded commands.", "website": "https://supabase.com/docs/guides/getting-started/mcp", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "API stability and uptime analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "query_execution": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "Query execution testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "schema_management": { "score": 85, @@ -52,7 +52,7 @@ } ], "methodology": "Schema management testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "type_generation": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Type generation testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "realtime_support": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Real-time functionality testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -92,14 +92,14 @@ "confidence": "high", "evidence": [ { - "source": "Supabase Auth", - "url": "https://supabase.com/docs/guides/auth", - "date": "2025-01-10", - "value": "Service role key with full database access, or anon key with RLS" + "source": "Supabase MCP Documentation", + "url": "https://supabase.com/docs/guides/getting-started/mcp", + "date": "2026-07-09", + "value": "Hosted server authenticates via OAuth 2.1 with dynamic client registration (browser consent flow) by default; personal access tokens are reserved for CI/CD and non-interactive use — the earlier service-role-key setup is no longer the documented path" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "row_level_security": { "score": 92, @@ -113,21 +113,21 @@ } ], "methodology": "RLS implementation review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "sql_injection_protection": { - "score": 78, - "confidence": "medium", + "score": 72, + "confidence": "high", "evidence": [ { - "source": "MCP SQL Execution", + "source": "Supabase MCP Security Documentation", "url": "https://supabase.com/docs/guides/getting-started/mcp", - "date": "2025-01-10", - "value": "AI can execute arbitrary SQL; requires careful prompt engineering" + "date": "2026-07-09", + "value": "2025 security research demonstrated prompt-injection attacks where instructions embedded in database rows caused agents with full SQL access to leak private data. Supabase responded with layered mitigations: SQL results wrapped with instructions discouraging the LLM from obeying embedded commands, read-only mode via a dedicated supabase_read_only_user, project scoping, and feature-group restrictions; manual tool-call approval is strongly recommended" } ], - "methodology": "SQL injection testing", - "last_verified": "2025-01-14" + "methodology": "Prompt-injection and SQL execution risk assessment; lowered from 78 because the attack class was demonstrated in practice — mitigations reduce but do not eliminate the risk when the agent reads untrusted data with write or broad read access", + "last_verified": "2026-07-09" }, "data_encryption": { "score": 85, @@ -141,21 +141,21 @@ } ], "methodology": "Encryption review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "key_management": { - "score": 75, + "score": 80, "confidence": "medium", "evidence": [ { - "source": "Supabase Keys", - "url": "https://supabase.com/docs/guides/api/api-keys", - "date": "2025-01-10", - "value": "Service role key has full access; key rotation available" + "source": "Supabase MCP Documentation", + "url": "https://supabase.com/docs/guides/getting-started/mcp", + "date": "2026-07-09", + "value": "OAuth-based hosted mode avoids long-lived keys in client config; PATs (revocable, org-scoped) are only needed for CI/CD; feature groups and project_ref scoping constrain what an authenticated session can touch" } ], - "methodology": "Key management review", - "last_verified": "2025-01-14" + "methodology": "Key management review; raised from 75 as the default auth path no longer places a full-access service role key in client configuration", + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data residency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "gdpr_compliance": { "score": 85, @@ -188,7 +188,7 @@ } ], "methodology": "GDPR compliance review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_exposure_to_llm": { "score": 70, @@ -202,7 +202,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "self_hosted_option": { "score": 90, @@ -216,7 +216,7 @@ } ], "methodology": "Self-hosting options review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -235,21 +235,21 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "open_source_transparency": { - "score": 95, + "score": 93, "confidence": "high", "evidence": [ { - "source": "Supabase GitHub", - "url": "https://github.com/supabase/supabase", - "date": "2025-01-10", - "value": "Fully open source under Apache 2.0 license, 75k+ stars" + "source": "supabase/mcp Repository", + "url": "https://github.com/supabase/mcp", + "date": "2026-07-09", + "value": "MCP server itself is fully open source under Apache 2.0 in the supabase/mcp monorepo (2.8k+ stars, 38 releases); the underlying Supabase platform is also open source" } ], - "methodology": "Source code review", - "last_verified": "2025-01-14" + "methodology": "Source code review; minor adjustment from 95 — previous evidence cited the platform repo rather than the MCP server, which remains pre-1.0", + "last_verified": "2026-07-09" }, "query_logging": { "score": 85, @@ -263,7 +263,7 @@ } ], "methodology": "Logging capabilities assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "community_activity": { "score": 90, @@ -277,7 +277,7 @@ } ], "methodology": "Community engagement analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -285,18 +285,18 @@ "overall_score": 86, "criteria": { "ease_of_setup": { - "score": 90, + "score": 92, "confidence": "high", "evidence": [ { "source": "Supabase MCP Setup", "url": "https://supabase.com/docs/guides/getting-started/mcp", - "date": "2025-01-10", - "value": "Simple setup with project URL and service key" + "date": "2026-07-09", + "value": "Hosted mode: add https://mcp.supabase.com/mcp?project_ref= (optionally &read_only=true) and complete the browser OAuth flow — no token creation needed; local CLI mode available at http://localhost:54321/mcp" } ], - "methodology": "Setup complexity assessment", - "last_verified": "2025-01-14" + "methodology": "Setup complexity assessment; small raise from 90 as the hosted OAuth flow removed manual key handling", + "last_verified": "2026-07-09" }, "developer_experience": { "score": 92, @@ -310,7 +310,7 @@ } ], "methodology": "Developer experience assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "scalability": { "score": 85, @@ -324,7 +324,7 @@ } ], "methodology": "Scalability testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "cost_efficiency": { "score": 82, @@ -338,7 +338,7 @@ } ], "methodology": "Cost analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "integration_ecosystem": { "score": 88, @@ -352,40 +352,42 @@ } ], "methodology": "Integration ecosystem review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } }, "strengths": [ - "Full PostgreSQL power with AI-assisted query building", - "Automatic TypeScript type generation from schema", - "Excellent row-level security for fine-grained access", - "Self-hosted option for complete data sovereignty", - "Outstanding developer experience and documentation", - "Fully open source with active community" + "Full PostgreSQL power with AI-assisted query building, migrations, and type generation", + "Hosted remote server with OAuth 2.1 dynamic client registration — no manual key handling", + "Layered security controls: read-only mode, project scoping, and feature-group tool restrictions", + "Prompt-injection mitigations wrap SQL results to discourage the LLM from obeying embedded commands", + "Development-branch workflow lets agents work against non-production data", + "Fully open source (Apache 2.0) with outstanding documentation and active community" ], "limitations": [ - "Service role key grants full database access", + "Demonstrated 2025 prompt-injection/data-leak attack class: instructions hidden in database rows can subvert agents given broad SQL access", "Query results and schema exposed to LLM provider", - "AI can execute potentially destructive SQL", - "Requires careful prompt engineering for safety", - "Real-time features add complexity", - "Connection pooling limits on free tier" + "AI can execute potentially destructive SQL unless read-only mode is enabled", + "Supabase recommends avoiding production data: use development branches, project scoping, and manual tool-call approval", + "Pre-1.0 server: breaking changes possible between versions", + "OAuth hosted mode has limited functionality for CLI/self-hosted environments" ], "metadata": { "license": "Apache 2.0", - "supported_platforms": ["All platforms with Node.js"], + "supported_platforms": ["Any MCP client (hosted remote); all platforms with Node.js (local)"], "programming_languages": ["TypeScript"], "mcp_version": "1.0", - "github_repo": "https://github.com/supabase/supabase", - "github_stars": 78000, - "api_dependency": "Supabase REST API / PostgreSQL", - "authentication": "Service Role Key or Anon Key", + "github_repo": "https://github.com/supabase/mcp", + "github_stars": 2800, + "remote_endpoint": "https://mcp.supabase.com/mcp (supports project_ref, read_only, and feature-group query parameters)", + "api_dependency": "Supabase Management API / PostgreSQL", + "authentication": "OAuth 2.1 with dynamic client registration (default); Personal Access Token for CI/CD", + "security_controls": ["read-only mode", "project scoping", "feature groups", "SQL result wrapping against prompt injection"], "first_release": "2025-01", "maintained_by": "Supabase", - "transport_types": ["stdio"], - "installation_methods": ["npm"] + "transport_types": ["streamable-http (hosted)", "stdio (local)"], + "installation_methods": ["Remote MCP endpoint", "npx @supabase/mcp-server-supabase (local)", "Supabase CLI local endpoint"] }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-tavily.json b/data/mcps/mcp-server-tavily.json index 1a53858..d7e5bc7 100644 --- a/data/mcps/mcp-server-tavily.json +++ b/data/mcps/mcp-server-tavily.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Tavily Server", "provider": "Tavily", - "version": "2025.2.0", - "last_evaluated": "2025-01-14", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "MCP server enabling AI models to perform real-time web search, content extraction, and web crawling. Designed specifically for AI agents with optimized search results, automatic content summarization, and source verification capabilities.", + "description": "Tavily's official MCP server (tavily-ai/tavily-mcp, MIT) enabling AI models to perform real-time web search, content extraction, site mapping, and web crawling. Available as a hosted remote endpoint at https://mcp.tavily.com/mcp/ (API key or OAuth) or locally via npx tavily-mcp. Designed specifically for AI agents with optimized search results and source verification capabilities.", "website": "https://tavily.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Search quality testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "response_latency": { "score": 85, @@ -38,7 +38,7 @@ } ], "methodology": "Latency benchmarking", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "content_extraction": { "score": 88, @@ -52,7 +52,7 @@ } ], "methodology": "Content extraction testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "result_freshness": { "score": 88, @@ -66,7 +66,7 @@ } ], "methodology": "Freshness testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 82, @@ -80,7 +80,7 @@ } ], "methodology": "Rate limiting behavior testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -92,14 +92,14 @@ "confidence": "high", "evidence": [ { - "source": "Tavily Authentication", - "url": "https://docs.tavily.com/", - "date": "2025-01-10", - "value": "API key authentication with standard security practices" + "source": "Tavily MCP Documentation", + "url": "https://docs.tavily.com/documentation/mcp", + "date": "2026-07-09", + "value": "API key authentication (env var locally; query parameter or header on the remote endpoint) plus OAuth support on the hosted MCP endpoint, recommended for production" } ], "methodology": "Authentication mechanism review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_handling": { "score": 78, @@ -113,7 +113,7 @@ } ], "methodology": "Data handling review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "content_filtering": { "score": 75, @@ -127,7 +127,7 @@ } ], "methodology": "Content filtering assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "source_verification": { "score": 80, @@ -141,7 +141,7 @@ } ], "methodology": "Source verification testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -160,7 +160,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "score": 75, @@ -174,7 +174,7 @@ } ], "methodology": "Data retention review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "third_party_sharing": { "score": 78, @@ -188,7 +188,7 @@ } ], "methodology": "Third-party sharing review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -207,7 +207,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "result_attribution": { "score": 90, @@ -221,7 +221,7 @@ } ], "methodology": "Attribution testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "api_transparency": { "score": 82, @@ -235,7 +235,7 @@ } ], "methodology": "API transparency review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "pricing_transparency": { "score": 85, @@ -243,13 +243,13 @@ "evidence": [ { "source": "Tavily Pricing", - "url": "https://tavily.com/pricing", - "date": "2025-01-10", - "value": "Clear pricing tiers with defined quotas" + "url": "https://www.tavily.com/pricing", + "date": "2026-07-09", + "value": "Clear credit-based pricing: 1,000 free API credits/month, paid plans from $30/month (Researcher) and $100/month (Startup), pay-as-you-go overage at $0.008/credit; basic search costs 1 credit, advanced search 2 credits" } ], "methodology": "Pricing review", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } }, @@ -261,14 +261,14 @@ "confidence": "high", "evidence": [ { - "source": "Tavily Quickstart", - "url": "https://docs.tavily.com/", - "date": "2025-01-10", - "value": "Simple API key setup with immediate access" + "source": "Tavily MCP Documentation", + "url": "https://docs.tavily.com/documentation/mcp", + "date": "2026-07-09", + "value": "Hosted remote endpoint at https://mcp.tavily.com/mcp/ requires only a URL plus API key (or OAuth); local install is a single npx tavily-mcp command" } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "api_reliability": { "score": 85, @@ -282,7 +282,7 @@ } ], "methodology": "Reliability assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "cost_efficiency": { "score": 82, @@ -290,13 +290,13 @@ "evidence": [ { "source": "Tavily Pricing", - "url": "https://tavily.com/pricing", - "date": "2025-01-10", - "value": "Free tier available; reasonable pricing for production" + "url": "https://www.tavily.com/pricing", + "date": "2026-07-09", + "value": "Free tier of 1,000 credits/month with no credit card; paid plans from $30/month with $0.008/credit pay-as-you-go overage" } ], "methodology": "Cost analysis", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "ai_optimization": { "score": 90, @@ -310,21 +310,21 @@ } ], "methodology": "AI integration testing", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" }, "sdk_support": { "score": 85, "confidence": "high", "evidence": [ { - "source": "Tavily SDKs", - "url": "https://docs.tavily.com/", - "date": "2025-01-10", - "value": "Python and JavaScript SDKs with MCP integration" + "source": "tavily-ai/tavily-mcp repository", + "url": "https://github.com/tavily-ai/tavily-mcp", + "date": "2026-07-09", + "value": "Official MCP server (npm tavily-mcp, v0.2.20) actively maintained alongside Python and JavaScript SDKs; repo tagline: production-ready MCP server with real-time search, extract, map & crawl" } ], "methodology": "SDK quality assessment", - "last_verified": "2025-01-14" + "last_verified": "2026-07-09" } } } @@ -332,31 +332,35 @@ "strengths": [ "Purpose-built for AI agents with optimized responses", "Real-time search with configurable time filtering", - "Automatic content extraction and summarization", + "Content extraction, site mapping, and crawling alongside search", "Source URLs provided for all results", - "Easy integration with MCP protocol", - "Good documentation and SDK support" + "Hosted remote MCP endpoint (mcp.tavily.com) with OAuth — no local install needed", + "Open-source MIT server (tavily-ai/tavily-mcp) with good documentation and SDK support" ], "limitations": [ "Search queries processed through third-party servers", "Rate limits based on pricing tier", "Web content quality varies by source", - "No self-hosted option available", + "Underlying search API is hosted-only — no self-hosted backend", "Query privacy depends on Tavily's policies", - "Costs can increase with heavy usage" + "Credit-based costs can increase with heavy usage (advanced search consumes 2 credits per call)" ], "metadata": { - "license": "Proprietary (API Service)", - "supported_platforms": ["All platforms with HTTP"], - "programming_languages": ["Python", "JavaScript", "TypeScript"], + "license": "MIT (MCP server); proprietary hosted API service", + "supported_platforms": ["All platforms with Node.js (local server); any MCP client (remote endpoint)"], + "programming_languages": ["TypeScript", "Python", "JavaScript"], "mcp_version": "1.0", "website": "https://tavily.com/", + "github_repo": "https://github.com/tavily-ai/tavily-mcp", + "package": "tavily-mcp", + "package_version": "0.2.20", + "remote_endpoint": "https://mcp.tavily.com/mcp/", "api_dependency": "Tavily Search API", - "authentication": "API Key", + "authentication": "API Key (env var or query parameter); OAuth on hosted endpoint", "first_release": "2024", "maintained_by": "Tavily", - "transport_types": ["stdio"], - "installation_methods": ["npm", "pip"] + "transport_types": ["stdio", "streamable-http"], + "installation_methods": ["npm", "npx", "remote-url"] }, "use_case_ratings": { "code-generation": { diff --git a/data/mcps/mcp-server-time.json b/data/mcps/mcp-server-time.json index 3929542..c5cad0a 100644 --- a/data/mcps/mcp-server-time.json +++ b/data/mcps/mcp-server-time.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "MCP Time Server", "provider": "Anthropic", - "version": "2025.9.25", - "last_evaluated": "2026-06-10", + "version": "2026.6.4", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Official MCP reference server for time and timezone operations. Provides AI models with current time information, timezone conversions, date calculations, and scheduling assistance. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest spec 2025-11-25).", + "description": "Official MCP reference server for time and timezone operations. Provides AI models with current time information, timezone conversions, date calculations, and scheduling assistance. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest stable spec 2025-11-25; 2026-07-28 spec revision at release-candidate stage as of 2026-07).", "website": "https://modelcontextprotocol.io/docs/servers/time", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Time accuracy verification", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "timezone_conversion_accuracy": { "score": 93, @@ -38,7 +38,7 @@ } ], "methodology": "Conversion accuracy testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "dst_handling": { "score": 90, @@ -52,7 +52,7 @@ } ], "methodology": "DST edge case testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "date_calculation_reliability": { "score": 91, @@ -66,7 +66,7 @@ } ], "methodology": "Date calculation testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "response_speed": { "score": 95, @@ -80,7 +80,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -99,7 +99,7 @@ } ], "methodology": "Operation authorization review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "system_time_manipulation_risk": { "score": 100, @@ -113,7 +113,7 @@ } ], "methodology": "System modification testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_exposure_risk": { "score": 92, @@ -127,7 +127,7 @@ } ], "methodology": "Data flow security analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "timezone_inference_privacy": { "score": 90, @@ -141,7 +141,7 @@ } ], "methodology": "Privacy impact assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "resource_consumption": { "score": 95, @@ -155,7 +155,7 @@ } ], "methodology": "Resource consumption testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Location privacy assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "temporal_pattern_exposure": { "score": 88, @@ -188,7 +188,7 @@ } ], "methodology": "Pattern analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "data_minimization": { "score": 92, @@ -202,7 +202,7 @@ } ], "methodology": "Data minimization review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Data sharing analysis", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 92, @@ -249,7 +249,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 95, @@ -263,7 +263,7 @@ } ], "methodology": "Source code review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "timezone_data_source_clarity": { "score": 88, @@ -277,7 +277,7 @@ } ], "methodology": "Data source transparency review", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "operation_performance": { "score": 95, @@ -310,7 +310,7 @@ } ], "methodology": "Performance benchmarking", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "reliability": { "score": 95, @@ -324,7 +324,7 @@ } ], "methodology": "Reliability testing", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 90, @@ -338,7 +338,7 @@ } ], "methodology": "Feature completeness assessment", - "last_verified": "2025-11-09" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 90, @@ -355,10 +355,16 @@ "url": "https://www.anthropic.com/news/donating-the-model-context-protocol-and-establishing-of-the-agentic-ai-foundation", "date": "2025-12-09", "value": "time is one of the seven reference servers still actively maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09" + }, + { + "source": "PyPI - mcp-server-time", + "url": "https://pypi.org/project/mcp-server-time/", + "date": "2026-07-09", + "value": "Latest release 2026.6.4 (published 2026-06-04); still listed among the seven maintained reference servers in modelcontextprotocol/servers README (repo ~88,300 stars, not archived)" } ], "methodology": "Community activity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } @@ -378,7 +384,7 @@ "Query patterns may reveal user schedule to LLM provider", "Depends on system time accuracy and timezone database updates", "No advanced scheduling or reminder capabilities", - "STATUS 2026-06-10: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); governance now under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" + "STATUS 2026-07-09: actively maintained reference server (one of 7 retained after the 2025-05-29 archival); latest PyPI release 2026.6.4 (2026-06-04); no known CVEs against this server; governance under the Agentic AI Foundation (Linux Foundation, 2025-12-09)" ], "metadata": { "license": "MIT", @@ -391,7 +397,7 @@ ], "mcp_version": "1.0", "github_repo": "https://github.com/modelcontextprotocol/servers", - "github_stars": 58700, + "github_stars": 88259, "api_dependency": "System time APIs, IANA Time Zone Database", "authentication": "None required", "first_release": "2024-11", diff --git a/data/mcps/mcp-server-vercel.json b/data/mcps/mcp-server-vercel.json index 140aa80..dc7bab7 100644 --- a/data/mcps/mcp-server-vercel.json +++ b/data/mcps/mcp-server-vercel.json @@ -3,11 +3,11 @@ "type": "mcp", "name": "Vercel MCP Server", "provider": "Vercel", - "version": "2026.6-beta", - "last_evaluated": "2026-06-10", + "version": "2026.7-beta", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Vercel's official hosted MCP server at mcp.vercel.com. Provides Vercel documentation search and tools to inspect teams, projects, deployments, and deployment logs. Remote-only with OAuth 2.1, a client allowlist, and mandatory consent; deliberately read-only at initial launch. Currently in Public Beta.", - "website": "https://vercel.com/docs/mcp/vercel-mcp", + "description": "Vercel's official hosted MCP server at mcp.vercel.com. Provides docs search plus tools to inspect teams, projects, deployments, build and runtime logs, and Agent Runs observability. No longer read-only: includes write-capable tools such as buy_domain (domain purchase), toolbar comment replies/edits/resolution, shareable-link creation for protected deployments, and CLI-guided deploys. Remote-only with OAuth 2.1, a client allowlist, and mandatory consent. Still in Beta on all plans.", + "website": "https://vercel.com/docs/agent-resources/vercel-mcp", "trust_vector": { "performance_reliability": { "overall_score": 83, @@ -18,13 +18,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Hosted on Vercel's own platform infrastructure with a single managed endpoint at https://mcp.vercel.com" } ], "methodology": "Endpoint stability analysis on Vercel platform infrastructure", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_success_rate": { "score": 84, @@ -32,13 +32,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Read-oriented tools (projects, deployments, logs) are thin wrappers over the stable Vercel REST API and succeed consistently within granted scopes" } ], "methodology": "Operation success testing across documented tools on real projects", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "search_accuracy": { "score": 85, @@ -46,13 +46,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Documentation search returns current Vercel docs, reducing hallucinated configuration answers in coding agents" } ], "methodology": "Relevance assessment of documentation search results for common platform queries", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 80, @@ -60,13 +60,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Returns structured errors for unauthorized scopes, unknown projects, and expired sessions; OAuth re-consent flow recovers expired grants" } ], "methodology": "Error handling testing across permission, scope, and session-expiry failures", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 78, @@ -74,18 +74,18 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Inherits Vercel API rate limits per authenticated user; MCP-specific limits are not separately published during Public Beta" } ], "methodology": "Rate limiting behavior observation; limited published detail during beta", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 87, + "overall_score": 80, "criteria": { "authentication_security": { "score": 92, @@ -93,27 +93,27 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "OAuth 2.1 authorization with mandatory user consent on every connection; no static API tokens are placed in client configuration" } ], "methodology": "Review of OAuth 2.1 flow, consent screens, and token lifecycle", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "scope_limitation": { - "score": 90, + "score": 70, "confidence": "high", "evidence": [ { - "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", - "date": "2026-06-10", - "value": "Deliberately read-only at initial launch: tools inspect teams, projects, deployments, and logs but cannot mutate or trigger deployments" + "source": "Vercel MCP Tools Reference", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp/tools", + "date": "2026-07-09", + "value": "The read-only launch posture has ended: current tools include buy_domain (purchases a domain with registrant details), reply_to_toolbar_thread / edit_toolbar_message / change_toolbar_thread_resolve_status, get_access_to_vercel_url (creates shareable links bypassing deployment protection), and deploy_to_vercel CLI guidance" } ], - "methodology": "Permission boundary testing of the shipped tool surface for write capability", - "last_verified": "2026-06-10" + "methodology": "Permission boundary testing of the shipped tool surface for write capability; lowered from 90 because write-capable tools, including one that spends money and one that mints protection-bypassing links, are now part of the default surface", + "last_verified": "2026-07-09" }, "token_exposure_risk": { "score": 85, @@ -121,27 +121,27 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Remote-only design means no locally stored long-lived secrets; OAuth tokens are short-lived and revocable from the Vercel dashboard" } ], "methodology": "Token storage and exposure-surface analysis for the remote-only model", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "unauthorized_action_risk": { - "score": 88, + "score": 74, "confidence": "high", "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", - "date": "2026-06-10", - "value": "A client allowlist restricts which MCP clients may connect, and the read-only launch surface means a confused or injected agent cannot modify infrastructure" + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", + "date": "2026-07-09", + "value": "Client allowlist and per-connection consent remain, with documented confused-deputy protection; however an injected agent could now purchase domains, create protection-bypassing shareable links, or post toolbar replies. Vercel's own docs recommend enabling human confirmation for tool execution" } ], - "methodology": "Threat modeling of agent misuse against the allowlisted, read-only tool surface", - "last_verified": "2026-06-10" + "methodology": "Threat modeling of agent misuse against the allowlisted tool surface; lowered from 88 following the addition of write-capable and spend-capable tools", + "last_verified": "2026-07-09" }, "action_auditability": { "score": 80, @@ -149,13 +149,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "OAuth grants are visible and revocable per user; access activity is attributable to the granting account, though a dedicated MCP audit log is not yet exposed" } ], "methodology": "Audit logging review of grant management and access attribution", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -168,13 +168,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Project metadata, deployment details, and log contents are returned to the LLM provider as tool results" } ], "methodology": "Data flow analysis of tool results from Vercel to LLM providers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 70, @@ -182,13 +182,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Deployment logs can contain secrets, tokens, or PII printed by application code; the server does not redact log contents before returning them" } ], "methodology": "Assessment of redaction controls on log-reading tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "organization_data_control": { "score": 82, @@ -202,7 +202,7 @@ } ], "methodology": "Access control review against Vercel team RBAC", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 76, @@ -216,7 +216,7 @@ } ], "methodology": "Analysis of downstream data sharing once tool results leave Vercel", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -229,13 +229,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Clear official documentation covering endpoint, OAuth setup, supported clients, tool capabilities, and security model" } ], "methodology": "Documentation completeness and accuracy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 82, @@ -243,13 +243,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Tool calls are visible in MCP client logs; connected integrations and grants are visible in the Vercel dashboard" } ], "methodology": "Logging and traceability assessment across client and dashboard", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 45, @@ -263,7 +263,7 @@ } ], "methodology": "Source availability and independent verifiability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_coverage_clarity": { "score": 84, @@ -271,18 +271,18 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", - "value": "Documented tool surface (docs search; team, project, deployment management views; deployment logs) matches observed behavior, with read-only status stated explicitly" + "value": "Tools reference (updated 2026-07-02) documents every tool with parameters, including the write-capable domain, toolbar, and access tools; documented surface matches observed behavior" } ], "methodology": "Comparison of documented tool surface against observed server capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "operational_excellence": { - "overall_score": 80, + "overall_score": 82, "criteria": { "ease_of_setup": { "score": 90, @@ -290,13 +290,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Zero installation: add https://mcp.vercel.com in a supported client and complete the OAuth consent flow" } ], "methodology": "Setup complexity assessment across allowlisted MCP clients", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 84, @@ -304,13 +304,13 @@ "evidence": [ { "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp", "date": "2026-06-10", "value": "Hosted on Vercel's edge-adjacent infrastructure; typical tool responses return quickly, with log retrieval slowest for large deployments" } ], "methodology": "Latency observation across representative tool calls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 80, @@ -324,21 +324,21 @@ } ], "methodology": "Uptime analysis combined with beta-status risk assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { - "score": 68, + "score": 78, "confidence": "high", "evidence": [ { - "source": "Vercel MCP Documentation", - "url": "https://vercel.com/docs/mcp/vercel-mcp", - "date": "2026-06-10", - "value": "Read-only launch scope excludes deployment triggering, environment variable management, and domain configuration; coverage is intentionally narrow" + "source": "Vercel MCP Tools Reference", + "url": "https://vercel.com/docs/agent-resources/vercel-mcp/tools", + "date": "2026-07-09", + "value": "Coverage now spans docs search, teams/projects/deployments, build and runtime logs (with filtering), eve Agent Runs observability and traces, domain availability checks and purchase, toolbar comment workflows, deployment URL fetching, and CLI-guided deploys; environment variable management is still absent" } ], - "methodology": "Feature completeness assessment against the full Vercel API surface", - "last_verified": "2026-06-10" + "methodology": "Feature completeness assessment against the full Vercel API surface; raised from 68 after substantial tool surface expansion (runtime logs, Agent Runs, domains, toolbar)", + "last_verified": "2026-07-09" }, "community_adoption": { "score": 78, @@ -352,24 +352,25 @@ } ], "methodology": "Adoption analysis across the Vercel developer ecosystem and MCP clients", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } }, "strengths": [ - "Notably strong security posture: OAuth 2.1, mandatory consent, client allowlist, and read-only launch scope", + "Strong authentication posture: OAuth 2.1, mandatory consent, client allowlist, and documented confused-deputy protection", "Remote-only design eliminates locally stored long-lived secrets", "Live documentation search reduces hallucinated Vercel configuration answers", - "Deployment log access enables fast AI-assisted debugging of failed builds", - "Zero-install setup in supported MCP clients", + "Build and runtime log access with rich filtering enables fast AI-assisted debugging", + "Agent Runs observability tools expose eve agent traces for production debugging", + "Zero-install setup in supported MCP clients, including one-command install via npx add-mcp", "Backed by Vercel's reliable platform infrastructure" ], "limitations": [ - "Read-only at launch: cannot trigger deployments or change configuration", + "No longer read-only: buy_domain can spend money, toolbar tools post/edit content, and get_access_to_vercel_url mints links that bypass deployment protection — human confirmation strongly recommended", "Closed source; implementation cannot be independently audited", - "Deployment logs may contain unredacted secrets or PII that flow to the LLM provider", - "Public Beta status means the tool surface and limits may change", + "Deployment and runtime logs may contain unredacted secrets or PII that flow to the LLM provider", + "Beta status means the tool surface and limits may change", "Client allowlist excludes unsupported or custom MCP clients", "No dedicated MCP-specific audit log exposed yet" ], @@ -377,16 +378,17 @@ "repository": "https://github.com/vercel/vercel-mcp-overview", "license": "Proprietary (closed source; public overview repo)", "maintained_by": "Vercel", - "status": "Public Beta", + "status": "Beta (all plans)", "remote_endpoint": "https://mcp.vercel.com", "authentication": "OAuth 2.1 with mandatory consent and client allowlist", "transport_types": [ "streamable-http (remote only)" ], "installation_methods": [ - "Remote MCP endpoint" + "Remote MCP endpoint", + "npx add-mcp https://mcp.vercel.com" ], - "write_access": "Read-only at initial launch by design", + "write_access": "Write-capable tools present since 2026 (domain purchase, toolbar comments, shareable links); no env var management", "mcp_version": "1.0" }, "use_case_ratings": { @@ -407,14 +409,14 @@ "notes": "Helps internal support diagnose customer-facing deployment issues from logs" }, "education": { - "overall": 72, - "notes": "Safe read-only surface is well suited to teaching deployment and platform concepts" + "overall": 70, + "notes": "Well suited to teaching deployment and platform concepts, though the surface is no longer purely read-only" } }, "best_for": [ "Developers debugging Vercel deployments and build failures with AI assistance", "Coding agents that need accurate, current Vercel documentation and project context", - "Teams that want platform visibility for agents without granting any write access", + "Teams that want platform and agent-run visibility for AI assistants with human confirmation enabled on write tools", "Security-conscious organizations evaluating remote MCP servers" ], "related_entities": [ diff --git a/data/mcps/mcp-server-zapier.json b/data/mcps/mcp-server-zapier.json index 9e80cc9..0489904 100644 --- a/data/mcps/mcp-server-zapier.json +++ b/data/mcps/mcp-server-zapier.json @@ -3,10 +3,10 @@ "type": "mcp", "name": "Zapier MCP Server", "provider": "Zapier", - "version": "2026.6", - "last_evaluated": "2026-06-10", + "version": "2026.7", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Zapier's hosted, proprietary MCP server that gives AI agents access to user-selected actions from 9,000+ connected apps (Gmail, Slack, Salesforce, and more, spanning tens of thousands of actions). Each user generates a personal remote server endpoint at mcp.zapier.com with per-app and per-action permissioning.", + "description": "Zapier's hosted, proprietary MCP server giving AI agents access to user-selected actions from 9,000+ connected apps (40,000+ actions). Users create servers at mcp.zapier.com with per-app and per-action permissioning; listed clients connect via OAuth, unlisted clients use a rotatable connection token sent as an Authorization Bearer header (recommended) or embedded in the server URL. Transport is Streamable HTTP; each tool call consumes two Zapier tasks from the plan quota.", "website": "https://zapier.com/mcp", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Platform stability and maturity analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "action_execution_success": { "score": 84, @@ -38,21 +38,21 @@ } ], "methodology": "Action execution success testing across common apps", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "app_integration_breadth": { "score": 95, "confidence": "high", "evidence": [ { - "source": "Zapier MCP", - "url": "https://zapier.com/mcp", - "date": "2026-06-10", - "value": "9,000+ apps and roughly 30,000-40,000 actions available, the broadest integration catalog of any MCP server" + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-07-09", + "value": "9,000+ app connections and 40,000+ actions available, the broadest integration catalog of any MCP server" } ], "methodology": "Integration catalog assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "rate_limit_handling": { "score": 80, @@ -61,12 +61,12 @@ { "source": "Zapier MCP Documentation", "url": "https://docs.zapier.com/mcp/home", - "date": "2026-06-10", - "value": "Plan-based usage limits on MCP tool calls; downstream app rate limits surface as action errors" + "date": "2026-07-09", + "value": "MCP is available on all Zapier plans with each tool call consuming two tasks from the plan quota; downstream app rate limits surface as action errors" } ], "methodology": "Rate and usage limit behavior review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "error_recovery": { "score": 78, @@ -80,40 +80,40 @@ } ], "methodology": "Failure mode and recovery testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, "security": { - "overall_score": 71, + "overall_score": 73, "criteria": { "authentication_security": { - "score": 70, + "score": 75, "confidence": "high", "evidence": [ { - "source": "Zapier MCP Documentation", - "url": "https://docs.zapier.com/mcp/home", - "date": "2026-06-10", - "value": "A user-specific server is generated at mcp.zapier.com; access is tied to that personal endpoint rather than a separately rotated credential in legacy URL-auth mode" + "source": "Zapier Help - Use Zapier MCP with your client", + "url": "https://help.zapier.com/hc/en-us/articles/36265392843917-Use-Zapier-MCP-with-your-client", + "date": "2026-07-09", + "value": "Listed clients authenticate via OAuth; unlisted clients use a connection token (shown once, rotatable, immediately invalidating the old token) sent as an Authorization Bearer header to https://mcp.zapier.com/api/v1/connect, with URL-embedded token retained as a fallback" } ], - "methodology": "Authentication mechanism review", - "last_verified": "2026-06-10" + "methodology": "Authentication mechanism review; score raised from 70 (2026-06-10) to reflect OAuth for listed clients and rotatable header-based bearer tokens replacing static URL-only auth", + "last_verified": "2026-07-09" }, "url_credential_exposure": { - "score": 58, + "score": 62, "confidence": "high", "evidence": [ { - "source": "Zapier MCP Documentation", - "url": "https://docs.zapier.com/mcp/home", - "date": "2026-06-10", - "value": "The per-user endpoint URL acts as a bearer credential: anyone who obtains it can invoke the user's enabled actions, so it must be treated as a secret and rotated if leaked" + "source": "Zapier Help - Use Zapier MCP with your client", + "url": "https://help.zapier.com/hc/en-us/articles/36265392843917-Use-Zapier-MCP-with-your-client", + "date": "2026-07-09", + "value": "Connection tokens still act as bearer credentials (anyone holding one can invoke the user's enabled actions), but Zapier now recommends header-based auth over URL-embedded tokens, shows tokens only once, and supports one-click rotation that immediately invalidates the old token" } ], - "methodology": "Credential exposure threat modeling of the endpoint URL", - "last_verified": "2026-06-10" + "methodology": "Credential exposure threat modeling; score raised from 58 (2026-06-10) for header-based auth guidance and immediate token rotation", + "last_verified": "2026-07-09" }, "blast_radius_control": { "score": 60, @@ -127,7 +127,7 @@ } ], "methodology": "Blast radius assessment across connected app categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "permission_granularity": { "score": 82, @@ -141,7 +141,7 @@ } ], "methodology": "Per-app and per-action permission model review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "credential_handling": { "score": 85, @@ -155,7 +155,7 @@ } ], "methodology": "Credential storage and isolation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -174,7 +174,7 @@ } ], "methodology": "Data flow analysis of action inputs and outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sensitive_data_protection": { "score": 75, @@ -188,7 +188,7 @@ } ], "methodology": "Data protection controls assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "third_party_data_sharing": { "score": 70, @@ -202,7 +202,7 @@ } ], "methodology": "Multi-party data sharing review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -216,7 +216,7 @@ } ], "methodology": "Compliance certification review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -235,7 +235,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "operation_visibility": { "score": 86, @@ -249,7 +249,7 @@ } ], "methodology": "Logging and traceability assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "open_source_transparency": { "score": 35, @@ -263,7 +263,7 @@ } ], "methodology": "Source availability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "tool_coverage_clarity": { "score": 83, @@ -277,7 +277,7 @@ } ], "methodology": "Tool surface documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } }, @@ -296,7 +296,7 @@ } ], "methodology": "Setup complexity assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_performance": { "score": 80, @@ -310,7 +310,7 @@ } ], "methodology": "Action latency characterization", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "reliability": { "score": 86, @@ -324,21 +324,21 @@ } ], "methodology": "Uptime and incident history analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "feature_coverage": { "score": 95, "confidence": "high", "evidence": [ { - "source": "Zapier MCP", - "url": "https://zapier.com/mcp", - "date": "2026-06-10", - "value": "Tens of thousands of actions across 9,000+ apps including Gmail, Slack, Salesforce, HubSpot, and Google Workspace" + "source": "Zapier MCP Documentation", + "url": "https://docs.zapier.com/mcp/home", + "date": "2026-07-09", + "value": "40,000+ actions across 9,000+ apps including Gmail, Slack, Salesforce, HubSpot, and Google Workspace" } ], "methodology": "Capability breadth assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "community_adoption": { "score": 84, @@ -352,13 +352,14 @@ } ], "methodology": "Adoption and ecosystem support analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } } } }, "strengths": [ - "Broadest action catalog of any MCP server: 9,000+ apps and tens of thousands of actions", + "Broadest action catalog of any MCP server: 9,000+ apps and 40,000+ actions", + "OAuth for listed clients and rotatable header-based connection tokens for unlisted clients", "Per-app and per-action permissioning enables least-privilege agent configuration", "Zero-install hosted setup with a personal endpoint generated at mcp.zapier.com", "Downstream app credentials held by Zapier under SOC 2 Type II audited controls, never exposed to the agent", @@ -367,10 +368,10 @@ ], "limitations": [ "Extremely broad blast radius by design: a prompt-injected agent can act across email, CRM, chat, and finance apps", - "The per-user endpoint URL acts as a bearer credential and must be treated as a secret", + "Connection tokens (and URL-embedded fallback) act as bearer credentials and must be treated as secrets, though rotation is one click", "Fully proprietary and hosted-only; no source inspection or self-hosting", "Business data from connected apps flows through Zapier's cloud and the LLM provider", - "Subject to plan-based usage limits and pricing", + "Each MCP tool call consumes two Zapier tasks, so costs scale with agent activity under plan-based quotas", "Expired downstream app connections require manual reauthorization" ], "metadata": { @@ -379,12 +380,13 @@ "Hosted remote server (any MCP client with remote support)" ], "api_dependency": "Zapier platform and 9,000+ downstream app APIs", - "authentication": "User-specific endpoint at mcp.zapier.com (URL acts as bearer credential)", + "authentication": "OAuth (listed clients); rotatable connection token via Authorization Bearer header or URL-embedded token (unlisted clients) at mcp.zapier.com", + "pricing": "Available on all Zapier plans; each MCP tool call consumes two Zapier tasks (last_verified 2026-07-09)", "compliance": "SOC 2 Type II", "maintained_by": "Zapier", "documentation": "https://docs.zapier.com/mcp/home", "transport_types": [ - "remote (hosted)" + "streamable-http (hosted)" ], "installation_methods": [ "hosted endpoint" diff --git a/data/models/claude-fable-5.json b/data/models/claude-fable-5.json index a04530d..447d2f8 100644 --- a/data/models/claude-fable-5.json +++ b/data/models/claude-fable-5.json @@ -4,13 +4,13 @@ "name": "Claude Fable 5", "provider": "Anthropic", "version": "20260609", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Anthropic's new top-tier model above Opus and the first generally available Mythos-class model. State-of-the-art on nearly all tested benchmarks at launch, including the highest frontier score on Cognition's FrontierCode. Adaptive thinking only, 1M context, 128K output.", + "description": "Anthropic's top-tier model above Opus and the most capable widely released Mythos-class model. State-of-the-art on nearly all tested benchmarks at launch, including the highest frontier score on Cognition's FrontierCode. Adaptive thinking only, 1M context, 128K output. Access was suspended globally 2026-06-12 under a US export-control directive after a reported safeguard bypass, and restored 2026-07-01 with a strengthened safety classifier.", "website": "https://www.anthropic.com/news/claude-fable-5-mythos-5", "trust_vector": { "performance_reliability": { - "overall_score": 98, + "overall_score": 96, "criteria": { "task_accuracy_code": { "score": 99, @@ -30,7 +30,7 @@ } ], "methodology": "Frontier coding benchmarks measuring real-world software engineering and long-horizon agentic coding tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 98, @@ -44,7 +44,7 @@ } ], "methodology": "Graduate and PhD-level reasoning benchmarks requiring multi-step problem solving, evaluated with adaptive thinking at high effort", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 97, @@ -58,7 +58,7 @@ } ], "methodology": "Comprehensive knowledge and multimodal testing across text and vision inputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 96, @@ -72,7 +72,7 @@ } ], "methodology": "Repeated-run consistency testing across effort levels; adaptive-thinking-only surface removes sampling variance controls", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "No manual thinking budgets or temperature/top_p sampling parameters; behavior steered via prompting and the effort parameter" }, "latency_p50": { @@ -87,7 +87,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes; limited launch-window sample", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "7.0s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile response time across diverse workloads; limited launch-window sample", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -115,27 +115,34 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { - "score": 99, + "score": 82, "confidence": "high", "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2026-06-10", - "value": "99.9%+ platform uptime (last 90 days); Fable 5 served on the same infrastructure" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days), but Fable 5 access was suspended entirely 2026-06-12 to 2026-07-01; elevated Fable 5 error incidents on 2026-07-03 post-redeployment" + }, + { + "source": "Anthropic: Redeploying Claude Fable 5", + "url": "https://www.anthropic.com/news/redeploying-fable-5", + "date": "2026-07-01", + "value": "Global access suspended 2026-06-12 under US export-control directive; controls lifted 2026-06-30 and access restored 2026-07-01" } ], - "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "methodology": "Historical uptime data from official status page plus availability-event review", + "last_verified": "2026-07-09", + "notes": "Score reduced from 99: the model was fully unavailable for ~19 days (2026-06-12 to 2026-07-01) during the export-control suspension" } }, - "notes": "Current highest-performing model in the registry. SOTA on nearly all tested benchmarks at launch, including the top frontier score on Cognition's FrontierCode. Latency data is preliminary (released 2026-06-09)." + "notes": "Current highest-performing model in the registry. SOTA on nearly all tested benchmarks at launch, including the top frontier score on Cognition's FrontierCode. Overall score reduced from 98 to 96 to reflect the 2026-06-12 to 2026-07-01 global access suspension (uptime criterion lowered); capability scores unchanged." }, "security": { - "overall_score": 92, + "overall_score": 90, "criteria": { "prompt_injection_resistance": { "score": 93, @@ -149,10 +156,10 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks; third-party red-team data still limited at launch", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { - "score": 95, + "score": 88, "confidence": "high", "evidence": [ { @@ -160,10 +167,17 @@ "url": "https://www.anthropic.com/news/claudes-constitution", "date": "2026-06-09", "value": "Constitutional AI alignment carried forward to the Mythos-class generation with strengthened refusal calibration" + }, + { + "source": "Anthropic: Redeploying Claude Fable 5", + "url": "https://www.anthropic.com/news/redeploying-fable-5", + "date": "2026-07-01", + "value": "Amazon researchers demonstrated a safeguard-bypass technique enabling the model to identify software vulnerabilities, triggering a US export-control suspension; a new safety classifier now blocks the reported jailbreak in over 99% of cases and routes blocked requests to Opus 4.8" } ], - "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2026-06-10" + "methodology": "Testing against adversarial prompt datasets and review of publicly disclosed bypass incidents", + "last_verified": "2026-07-09", + "notes": "Score reduced from 95 after the publicly confirmed June 2026 safeguard bypass; mitigations (layered classifier, defense in depth) deployed before the 2026-07-01 redeployment" }, "data_leakage_prevention": { "score": 88, @@ -177,7 +191,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 96, @@ -187,11 +201,11 @@ "source": "Anthropic Launch Announcement", "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", "date": "2026-06-09", - "value": "Released under Anthropic's Responsible Scaling Policy with frontier-tier safeguards; Claude Mythos 5 itself restricted to research partners" + "value": "Released under Anthropic's Responsible Scaling Policy with frontier-tier safeguards; Claude Mythos 5 offered separately via invitation-only Project Glasswing for defensive cybersecurity workflows" } ], "methodology": "Comprehensive safety testing across harmful content categories per Responsible Scaling Policy", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 89, @@ -205,10 +219,10 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Frontier-tier safety posture; the unrestricted Mythos-class research model (Claude Mythos 5) is limited to research partners while Fable 5 is the generally available variant. Independent red-team coverage still accumulating at launch." + "notes": "Frontier-tier safety posture, tested in practice: a safeguard bypass found by Amazon researchers led to a US export-control suspension (2026-06-12 to 2026-07-01), answered with a strengthened layered safety classifier before redeployment. Claude Mythos 5 is offered separately via invitation-only Project Glasswing. Overall score reduced from 92 to 90 to reflect the confirmed bypass and its mitigation." }, "privacy_compliance": { "overall_score": 93, @@ -225,7 +239,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -239,7 +253,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Configurable; zero-retention available", @@ -253,7 +267,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -267,7 +281,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -281,7 +295,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -295,7 +309,7 @@ } ], "methodology": "Review of data handling practices and trust center documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Same strong Anthropic compliance posture as the Opus line: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API traffic." @@ -315,7 +329,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Thinking text is omitted by default; opt in to summarized display for visible reasoning" }, "hallucination_rate": { @@ -330,7 +344,7 @@ } ], "methodology": "Testing on factual QA datasets; independent measurement still limited at launch", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -344,7 +358,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 89, @@ -358,7 +372,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 93, @@ -372,7 +386,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -386,7 +400,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 96, @@ -400,7 +414,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong documentation and guardrails. Thinking content is omitted by default (summarized display is opt-in), which slightly reduces out-of-the-box reasoning visibility compared to older Opus defaults." @@ -420,7 +434,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "One new breaking change vs Opus 4.8: explicit thinking disabled returns 400 \u2014 omit the thinking parameter instead. No temperature/top_p sampling parameters." }, "sdk_quality": { @@ -435,7 +449,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -449,7 +463,7 @@ } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -463,7 +477,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -477,22 +491,22 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { - "score": 88, - "confidence": "medium", + "score": 90, + "confidence": "high", "evidence": [ { - "source": "Anthropic Launch Announcement", - "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5", - "date": "2026-06-09", - "value": "Available on the Anthropic API at launch; first generally available Mythos-class model \u2014 cloud-provider rollout following" + "source": "Anthropic Models Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-07-09", + "value": "Generally available on the Claude API, Claude Platform on AWS, Amazon Bedrock, Google Cloud Vertex AI, and Microsoft Foundry; also on Claude.ai, Claude Code, and Claude Cowork since 2026-07-01" } ], "methodology": "Analysis of third-party integrations and availability surfaces", - "last_verified": "2026-06-10", - "notes": "Released 2026-06-09; multi-cloud availability narrower than Opus line at evaluation time" + "last_verified": "2026-07-09", + "notes": "Multi-cloud availability now matches the Opus line; availability was interrupted 2026-06-12 to 2026-07-01 by the export-control suspension" }, "license_terms": { "score": 92, @@ -506,10 +520,10 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Same API surface as Opus 4.7/4.8 makes adoption straightforward for existing Claude users. Day-old release means ecosystem and operational track record are still maturing." + "notes": "Same API surface as Opus 4.7/4.8 makes adoption straightforward for existing Claude users. Now generally available across all major cloud platforms; operational track record includes the 2026-06-12 to 2026-07-01 suspension and post-redeployment error spikes in early July 2026." } }, "use_case_ratings": { @@ -597,7 +611,7 @@ "strengths": [ "State-of-the-art on nearly all tested benchmarks at launch; highest-performing model in the registry", "Highest frontier score on Cognition's FrontierCode coding benchmark", - "First generally available Mythos-class model (Claude Mythos 5 itself is restricted to research partners)", + "Most capable widely released Mythos-class model (Claude Mythos 5 is invitation-only via Project Glasswing)", "1M token context window with 128K max output, text + vision", "Adaptive thinking with effort parameter (low/medium/high/xhigh/max) for cost/quality control", "Strong compliance posture: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API" @@ -606,7 +620,9 @@ "Premium pricing at $10/$50 per 1M tokens (2x Opus 4.8)", "Adaptive thinking only \u2014 no manual thinking budgets, and no temperature/top_p sampling parameters", "Explicit thinking-disabled requests return 400 (omit the thinking parameter instead)", - "Released 2026-06-09 \u2014 independent benchmark replication and operational track record still limited", + "Released 2026-06-09 \u2014 independent benchmark replication still limited", + "Access suspended globally 2026-06-12 to 2026-07-01 under a US export-control directive following a reported safeguard bypass; restored with a strengthened safety classifier that routes blocked requests to Opus 4.8", + "Elevated error incidents in early July 2026 following redeployment (status page)", "Higher latency than Sonnet/Haiku tiers, especially at xhigh/max effort" ], "best_for": [ @@ -625,8 +641,8 @@ "pricing": { "input": "$10.00 per 1M tokens", "output": "$50.00 per 1M tokens", - "notes": "2x Opus 4.8 pricing. Batch API 50% discount and prompt caching savings apply.", - "last_verified": "2026-06-10" + "notes": "2x Opus 4.8 pricing, unchanged since launch. Batch API 50% discount and prompt caching savings apply. On Claude.ai paid plans, included at up to 50% of weekly usage limits through 2026-07-07, then usage-credit pricing.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 128000, @@ -654,7 +670,7 @@ "open_source": false, "architecture": "Mythos-class transformer with Constitutional AI alignment; adaptive thinking only with effort parameter (low/medium/high/xhigh/max)", "parameters": "Not disclosed", - "knowledge_cutoff": "Not disclosed", + "knowledge_cutoff": "January 2026 (reliable knowledge and training data cutoff)", "release_date": "2026-06-09" }, "related_entities": [ diff --git a/data/models/claude-haiku-4-5.json b/data/models/claude-haiku-4-5.json index abdd0fd..459d518 100644 --- a/data/models/claude-haiku-4-5.json +++ b/data/models/claude-haiku-4-5.json @@ -3,10 +3,10 @@ "type": "model", "name": "Claude Haiku 4.5", "provider": "Anthropic", - "version": "20251015", - "last_evaluated": "2025-11-17", + "version": "20251001", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Anthropic's fastest model released October 2025. Best coding performance (73.3% SWE-bench) at 1/3 cost and 2x speed of Sonnet 4. First Haiku with extended thinking.", + "description": "Anthropic's fastest model, released October 2025 and still the speed tier of the current lineup (Active as of 2026-07). At launch it beat the since-retired Sonnet 4 on coding (73.3% vs 72.7% SWE-bench) at 1/3 the cost and 2x the speed. First Haiku with extended thinking.", "website": "https://www.anthropic.com/claude", "trust_vector": { @@ -31,7 +31,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 89, @@ -45,7 +45,7 @@ } ], "methodology": "Reasoning benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 90, @@ -59,7 +59,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 88, @@ -73,7 +73,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.4s", @@ -87,7 +87,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "0.9s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "context_window": { "value": "200,000 tokens", @@ -115,7 +115,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -123,13 +123,13 @@ "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2025-10-20", - "value": "99.95% uptime" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days)" } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Best coding model for the price. 73.3% SWE-bench beats Sonnet 4. 4-5x faster than Sonnet 4.5. Flagship performance at budget pricing." @@ -150,7 +150,7 @@ } ], "methodology": "OWASP LLM01 testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -164,7 +164,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 84, @@ -178,7 +178,7 @@ } ], "methodology": "Privacy policy analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_safety": { "score": 89, @@ -192,7 +192,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "api_security": { "score": 86, @@ -206,7 +206,7 @@ } ], "methodology": "Security review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Strong security with ASL-2 standard. Constitutional AI guardrails. Less restrictive than ASL-3 models." @@ -227,7 +227,7 @@ } ], "methodology": "Enterprise docs review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -241,7 +241,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -255,7 +255,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 87, @@ -269,7 +269,7 @@ } ], "methodology": "Data protection review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 92, @@ -283,7 +283,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -297,7 +297,7 @@ } ], "methodology": "Data handling review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Excellent privacy with ephemeral data. HIPAA eligible at budget pricing. Strong compliance." @@ -318,7 +318,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 83, @@ -332,7 +332,7 @@ } ], "methodology": "QA testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 82, @@ -346,7 +346,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 84, @@ -360,7 +360,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 89, @@ -374,7 +374,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -388,7 +388,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "guardrails": { "score": 91, @@ -402,7 +402,7 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with extended thinking (first for Haiku). Strong guardrails. Latest training cutoff (Feb 2025)." @@ -423,7 +423,7 @@ } ], "methodology": "API design review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -437,7 +437,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 89, @@ -448,10 +448,16 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2025-10-15", "value": "6-month deprecation" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "claude-haiku-4-5-20251001 Active; tentative retirement not sooner than October 15, 2026; listed in the current lineup as the fastest model" } ], "methodology": "Policy review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 88, @@ -465,7 +471,7 @@ } ], "methodology": "Monitoring tools review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "support_quality": { "score": 90, @@ -479,7 +485,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 91, @@ -493,7 +499,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -507,7 +513,7 @@ } ], "methodology": "License review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Excellent for high-volume production. Budget pricing enables multi-agent architectures. Fast deployment." @@ -517,53 +523,53 @@ "use_case_ratings": { "code-generation": { "overall": 96, - "notes": "World's best coding model for the price. 73.3% SWE-bench beats even Sonnet 4. Perfect for high-volume coding.", - "alternatives": ["claude-sonnet-4-5", "claude-opus-4"] + "notes": "Best coding model for the price at launch: 73.3% SWE-bench beat the since-retired Sonnet 4. Perfect for high-volume coding.", + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "customer-support": { "overall": 93, "notes": "Exceptional for customer support. Flagship performance at $1/$5. 0.4s latency perfect for chat.", - "alternatives": ["claude-sonnet-4", "gpt-4o-mini"] + "alternatives": ["claude-sonnet-4-6", "gpt-4o-mini"] }, "content-creation": { "overall": 87, "notes": "Strong content creation at budget pricing. Fast generation (4-5x faster than Sonnet 4.5).", - "alternatives": ["claude-opus-4", "claude-sonnet-4-5"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "data-analysis": { "overall": 85, "notes": "Good analytical capabilities with extended thinking. Great value for cost.", - "alternatives": ["claude-opus-4", "claude-sonnet-4"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "research-assistant": { "overall": 89, "notes": "Excellent for research. 200K context, extended thinking, fast, budget pricing.", - "alternatives": ["claude-opus-4", "gpt-5"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "legal-compliance": { "overall": 88, "notes": "HIPAA eligible with strong privacy at budget pricing. Good for legal document processing.", - "alternatives": ["claude-opus-4", "claude-sonnet-4"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "healthcare": { "overall": 87, "notes": "HIPAA eligible. Budget pricing enables high-volume clinical documentation.", - "alternatives": ["claude-opus-4", "claude-opus-4-1"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "financial-analysis": { "overall": 84, "notes": "Good for standard financial analysis. Extended thinking helps with modeling.", - "alternatives": ["claude-opus-4", "openai-o3"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] }, "education": { "overall": 90, "notes": "Exceptional for education. Fast, affordable, strong explanations. Perfect for high-volume use.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] }, "creative-writing": { "overall": 86, "notes": "Strong creative writing. Fast generation good for drafting. Budget pricing for iteration.", - "alternatives": ["claude-opus-4", "gpt-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] } }, @@ -573,15 +579,16 @@ "Budget pricing: $1/$5 per 1M tokens (1/3 cost of Sonnet 4)", "First Haiku with extended thinking, computer use, context awareness", "HIPAA eligible with ephemeral data handling", - "200K context, 64K output (more than Opus 4)", - "Latest training cutoff: February 2025" + "200K context, 64K output", + "Still Active in the current lineup as of 2026-07 (tentative retirement not sooner than 2026-10-15)" ], "limitations": [ "ASL-2 security (less restrictive than ASL-3 Opus/Sonnet models)", - "Slightly lower reasoning than Opus 4 for complex tasks", - "Training cutoff February 2025", - "New model with less battle-testing than predecessors" + "Lower reasoning ceiling than Sonnet/Opus tiers for complex tasks", + "Reliable knowledge cutoff February 2025 — oldest in the current Claude lineup", + "200K context vs 1M on Sonnet 4.6/Sonnet 5 and current Opus models", + "No adaptive thinking or effort parameter (extended thinking only)" ], "best_for": [ @@ -593,17 +600,18 @@ ], "not_recommended_for": [ - "Mission-critical tasks requiring absolute maximum reasoning (use Opus 4)", - "Tasks requiring latest training data (post-February 2025)", - "Scenarios requiring ASL-3 security level" + "Mission-critical tasks requiring absolute maximum reasoning (use Opus 4.8 or Claude Fable 5)", + "Tasks requiring recent knowledge (reliable cutoff February 2025)", + "Scenarios requiring ASL-3 security level", + "Contexts beyond 200K tokens" ], "metadata": { "pricing": { "input": "$1.00 per 1M tokens", "output": "$5.00 per 1M tokens", - "notes": "Best value in flagship class. Batch API 50% discount ($1/$2.50). Prompt caching up to 90% savings.", - "last_verified": "2025-11-17" + "notes": "Best value in flagship class. Batch API 50% discount ($1/$2.50). Prompt caching up to 90% savings. Confirmed unchanged at $1/$5 as of 2026-07-09.", + "last_verified": "2026-07-09" }, "context_window": 200000, "max_output_tokens": 64000, @@ -627,11 +635,11 @@ "open_source": false, "architecture": "Transformer-based with Constitutional AI and extended thinking", "parameters": "Not disclosed", - "training_cutoff": "February 2025", + "training_cutoff": "July 2025 (reliable knowledge through February 2025)", "safety_level": "ASL-2" }, - "related_entities": ["claude-sonnet-4-5", "claude-sonnet-4", "claude-opus-4", "gpt-4o-mini"], + "related_entities": ["claude-sonnet-4-6", "claude-sonnet-4-5", "claude-opus-4-8", "gpt-4o-mini"], "tags": [ "coding", diff --git a/data/models/claude-opus-4-1.json b/data/models/claude-opus-4-1.json index a75406a..3bb2cc7 100644 --- a/data/models/claude-opus-4-1.json +++ b/data/models/claude-opus-4-1.json @@ -3,10 +3,10 @@ "type": "model", "name": "Claude Opus 4.1", "provider": "Anthropic", - "version": "20250530", - "last_evaluated": "2025-11-07", + "version": "20250805", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Anthropic's most powerful model with state-of-the-art reasoning, ASL-3 safety level, and exceptional performance on complex tasks. Flagship model for mission-critical applications.", + "description": "DEPRECATED: Anthropic deprecated Claude Opus 4.1 (claude-opus-4-1-20250805) on 2026-06-05; it will be retired on 2026-08-05 and requests will then fail. Recommended replacement: Claude Opus 4.8. Historically a flagship model with state-of-the-art reasoning, ASL-3 safety level, and exceptional performance on complex tasks.", "website": "https://www.anthropic.com/claude", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Coding benchmarks and real-world engineering tasks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 98, @@ -45,7 +45,7 @@ } ], "methodology": "PhD-level reasoning and mathematics benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 95, @@ -59,7 +59,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 94, @@ -73,7 +73,7 @@ } ], "methodology": "Internal consistency testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "2.1s", @@ -87,7 +87,7 @@ } ], "methodology": "Real-world API latency measurements", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "4.2s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile measurements", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "context_window": { "value": "200,000 tokens", @@ -115,21 +115,21 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, "confidence": "high", "evidence": [ { - "source": "Anthropic Status", - "url": "https://status.anthropic.com/", - "date": "2025-11-01", - "value": "99.95% uptime (last 90 days)" + "source": "Anthropic Status Page", + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days); model still served until retirement on 2026-08-05" } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Highest reasoning capability among all models. Best for extremely complex, mission-critical tasks requiring maximum intelligence." @@ -150,7 +150,7 @@ } ], "methodology": "OWASP LLM security testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 95, @@ -164,7 +164,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 88, @@ -178,7 +178,7 @@ } ], "methodology": "Privacy policy and data handling review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_safety": { "score": 96, @@ -192,7 +192,7 @@ } ], "methodology": "Safety testing across harmful content", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "api_security": { "score": 90, @@ -206,7 +206,7 @@ } ], "methodology": "API security feature review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Industry-leading security with ASL-3 safety classification. Best-in-class for high-risk applications." @@ -227,7 +227,7 @@ } ], "methodology": "Enterprise documentation review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -241,7 +241,7 @@ } ], "methodology": "Privacy policy analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -255,7 +255,7 @@ } ], "methodology": "Terms of service review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 90, @@ -269,7 +269,7 @@ } ], "methodology": "Data protection capabilities review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -283,7 +283,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 98, @@ -297,7 +297,7 @@ } ], "methodology": "Data handling practices review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy with zero retention and HIPAA eligibility. Best for highly regulated industries." @@ -318,7 +318,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 89, @@ -332,7 +332,7 @@ } ], "methodology": "Factual accuracy testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -346,7 +346,7 @@ } ], "methodology": "Bias benchmarks and testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 88, @@ -360,7 +360,7 @@ } ], "methodology": "Qualitative confidence assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 93, @@ -374,7 +374,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -388,7 +388,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "guardrails": { "score": 97, @@ -402,14 +402,14 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Excellent transparency with superior explainability. ASL-3 classification demonstrates commitment to safety and transparency." }, "operational_excellence": { - "overall_score": 91, + "overall_score": 86, "criteria": { "api_design_quality": { "score": 93, @@ -423,7 +423,7 @@ } ], "methodology": "API design review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -437,10 +437,10 @@ } ], "methodology": "SDK quality assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 90, + "score": 75, "confidence": "high", "evidence": [ { @@ -448,10 +448,17 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2025-06-01", "value": "Clear versioning with deprecation notices" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "Deprecated 2026-06-05; retirement 2026-08-05; recommended replacement claude-opus-4-8" } ], "methodology": "Versioning policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09", + "notes": "Score reduced from 90: model is deprecated with retirement imminent (2026-08-05)" }, "monitoring_observability": { "score": 88, @@ -465,7 +472,7 @@ } ], "methodology": "Observability tools review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "support_quality": { "score": 92, @@ -479,10 +486,10 @@ } ], "methodology": "Support quality assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { - "score": 91, + "score": 84, "confidence": "high", "evidence": [ { @@ -493,7 +500,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "license_terms": { "score": 93, @@ -507,63 +514,63 @@ } ], "methodology": "License terms review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, - "notes": "Strong operational maturity with enterprise-grade support and documentation. Well-suited for mission-critical applications." + "notes": "Deprecated 2026-06-05 with retirement on 2026-08-05; migration target is Claude Opus 4.8 ($5/$25, a 67% price cut vs Opus 4.1's $15/$75). Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { "code-generation": { "overall": 95, - "notes": "Exceptional for complex software architecture and system design. Best for mission-critical code requiring maximum reliability.", - "alternatives": ["claude-sonnet-4-5", "gpt-5"] + "notes": "Exceptional for complex software architecture and system design in its era. Deprecated — migrate to Opus 4.8.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "customer-support": { "overall": 90, "notes": "Excellent quality but higher latency and cost than alternatives. Best for premium support requiring maximum empathy.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] }, "content-creation": { "overall": 96, "notes": "Outstanding for long-form, complex content requiring deep thinking. Natural, engaging writing.", - "alternatives": ["gpt-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] }, "data-analysis": { "overall": 97, - "notes": "Best-in-class for complex analytical tasks. Exceptional at multi-step reasoning and insight generation.", - "alternatives": ["claude-sonnet-4-5"] + "notes": "Best-in-class for complex analytical tasks in its era. Exceptional at multi-step reasoning and insight generation.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "research-assistant": { "overall": 97, "notes": "Superior for academic and professional research. Exceptional synthesis and critical analysis.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] }, "legal-compliance": { "overall": 95, - "notes": "Best for legal work requiring maximum accuracy and privacy. HIPAA eligible with zero retention.", - "alternatives": ["claude-sonnet-4-5"] + "notes": "Best for legal work requiring maximum accuracy and privacy in its era. HIPAA eligible with zero retention.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "healthcare": { "overall": 94, - "notes": "Top choice for healthcare with HIPAA eligibility and ASL-3 safety. Maximum privacy and accuracy.", - "alternatives": ["claude-sonnet-4-5"] + "notes": "Top choice for healthcare in its era with HIPAA eligibility and ASL-3 safety. Maximum privacy and accuracy.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "financial-analysis": { "overall": 96, "notes": "Exceptional for complex financial modeling and risk analysis. Superior quantitative reasoning.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] }, "education": { "overall": 95, "notes": "Outstanding for advanced education with detailed explanations and Socratic teaching.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] }, "creative-writing": { "overall": 94, "notes": "Excellent for sophisticated creative projects. Strong narrative structure and character depth.", - "alternatives": ["gpt-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] } }, @@ -577,6 +584,7 @@ ], "limitations": [ + "DEPRECATED 2026-06-05; retires 2026-08-05 — migrate to Claude Opus 4.8 (claude-opus-4-8) before then", "Highest latency (~2.1s p50) and cost among evaluated models", "Premium pricing ($15/$75 per 1M tokens)", "Overkill for simple tasks - use Sonnet for better value", @@ -593,6 +601,7 @@ ], "not_recommended_for": [ + "Any new deployments — the model is deprecated and retires 2026-08-05; use Claude Opus 4.8", "Simple tasks where Sonnet provides better value", "Real-time applications requiring sub-second latency", "High-volume, cost-sensitive workloads", @@ -603,10 +612,11 @@ "pricing": { "input": "$15.00 per 1M tokens", "output": "$75.00 per 1M tokens", - "notes": "Premium tier - 5x cost of Sonnet, use only when necessary", - "last_verified": "2025-11-09" + "notes": "Premium tier - 5x cost of Sonnet. Confirmed still $15/$75 as of 2026-07-09; successor Opus 4.8 costs $5/$25, so migration also cuts cost 67%.", + "last_verified": "2026-07-09" }, "context_window": 200000, + "max_output": 32000, "languages": [ "English", "Spanish", @@ -627,10 +637,10 @@ "parameters": "Not disclosed" }, - "related_entities": ["claude-sonnet-4-5", "gpt-5", "gemini-2-5-pro"], + "related_entities": ["claude-opus-4-8", "claude-opus-4-5", "claude-sonnet-4-6", "gpt-5-5"], "tags": [ - "flagship", + "deprecated", "highest-reasoning", "asl-3-safety", "hipaa-eligible", diff --git a/data/models/claude-opus-4-5.json b/data/models/claude-opus-4-5.json index 8f87e7b..46c74fe 100644 --- a/data/models/claude-opus-4-5.json +++ b/data/models/claude-opus-4-5.json @@ -4,7 +4,7 @@ "name": "Claude Opus 4.5", "provider": "Anthropic", "version": "20251101", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "SUPERSEDED: no longer Anthropic's most capable model — succeeded by Opus 4.6, 4.7, 4.8 (2026-05-28) and Claude Fable 5 (2026-06-09, new top tier). At launch it scored 80.9% SWE-bench and was the first model to exceed 80% on SWE-bench Verified, with a unique effort parameter for compute control.", "website": "https://www.anthropic.com/claude/opus", @@ -37,7 +37,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring real-world software engineering tasks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 95, @@ -57,7 +57,7 @@ } ], "methodology": "Graduate and PhD-level reasoning benchmarks requiring multi-step problem solving", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 95, @@ -77,7 +77,7 @@ } ], "methodology": "Comprehensive knowledge and multimodal testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 94, @@ -91,7 +91,7 @@ } ], "methodology": "Internal testing with effort parameter across quality levels", - "last_verified": "2026-01-14", + "last_verified": "2026-07-09", "notes": "Effort parameter allows precise control over output quality and consistency" }, "latency_p50": { @@ -106,7 +106,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "5.0s", @@ -120,7 +120,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "context_window": { "value": "200,000 tokens", @@ -134,7 +134,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -142,13 +142,13 @@ "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2026-01-01", - "value": "99.95% uptime (last 90 days)" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days)" } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Industry-leading coding capabilities with 80.9% SWE-bench. Unique effort parameter allows compute control. Exceptional abstract reasoning (37.6% ARC-AGI-2)." @@ -169,7 +169,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 94, @@ -183,7 +183,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 87, @@ -197,7 +197,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_safety": { "score": 95, @@ -211,7 +211,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "api_security": { "score": 88, @@ -225,7 +225,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strongest safety posture in the Claude family. Enhanced Constitutional AI provides industry-leading jailbreak resistance." @@ -246,7 +246,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -260,7 +260,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -274,7 +274,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -288,7 +288,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -302,7 +302,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -316,7 +316,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy posture with ephemeral data handling and strong compliance certifications. HIPAA eligible for healthcare." @@ -337,7 +337,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -351,7 +351,7 @@ } ], "methodology": "Testing on factual QA datasets and real-world usage", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -365,7 +365,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 87, @@ -379,7 +379,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -393,7 +393,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -407,7 +407,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "guardrails": { "score": 96, @@ -421,7 +421,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strong explainability with effort parameter control. Enhanced Constitutional AI provides transparency in alignment approach." @@ -442,7 +442,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -456,7 +456,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -473,10 +473,16 @@ "url": "https://www.anthropic.com/news/claude-opus-4-8", "date": "2026-06-10", "value": "Opus 4.5 superseded by Opus 4.6 (2026-02-05), 4.7 (2026-04-16), 4.8 (2026-05-28); Claude Fable 5 (2026-06-09) is the new top tier" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "claude-opus-4-5-20251101 still Active (not deprecated); tentative retirement not sooner than November 24, 2026" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -490,7 +496,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -504,7 +510,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 92, @@ -518,7 +524,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -532,7 +538,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Excellent operational maturity with multi-cloud availability. Effort parameter adds unique control capability. Enterprise-ready." @@ -608,7 +614,8 @@ "Premium pricing ($5/$25 per 1M tokens)", "No native audio capabilities", "Training data transparency limited (industry standard)", - "SUPERSEDED: Opus 4.6/4.7/4.8 and Claude Fable 5 (2026-06-09) are newer; no longer Anthropic's most capable model" + "SUPERSEDED: Opus 4.6/4.7/4.8 and Claude Fable 5 (2026-06-09) are newer; no longer Anthropic's most capable model", + "Listed as a legacy model in Anthropic's docs; still Active on the API with tentative retirement not sooner than 2026-11-24" ], "best_for": [ @@ -630,8 +637,8 @@ "pricing": { "input": "$5.00 per 1M tokens", "output": "$25.00 per 1M tokens", - "notes": "67% reduction from Opus 4.1. Batch API 50% discount. Prompt caching up to 90% savings.", - "last_verified": "2026-01-14" + "notes": "67% reduction from Opus 4.1. Batch API 50% discount. Prompt caching up to 90% savings. Confirmed unchanged at $5/$25 as of 2026-07-09.", + "last_verified": "2026-07-09" }, "context_window": 200000, "max_output": 64000, diff --git a/data/models/claude-opus-4-6.json b/data/models/claude-opus-4-6.json index a7d4ea4..f26bb91 100644 --- a/data/models/claude-opus-4-6.json +++ b/data/models/claude-opus-4-6.json @@ -4,7 +4,7 @@ "name": "Claude Opus 4.6", "provider": "Anthropic", "version": "20260205", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Anthropic's frontier Opus released February 2026 with 80.8% SWE-bench Verified, breakthrough 68.8% ARC-AGI-2 abstract reasoning, adaptive thinking, and a 1M token context window. Now two generations behind Opus 4.8 but still served.", "website": "https://www.anthropic.com/claude/opus", @@ -37,7 +37,7 @@ } ], "methodology": "Industry-standard coding and agentic benchmarks measuring real-world software engineering and computer-use tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 97, @@ -57,7 +57,7 @@ } ], "methodology": "Abstract reasoning and multi-step problem solving benchmarks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 95, @@ -71,7 +71,7 @@ } ], "methodology": "Comprehensive knowledge and multimodal testing across published benchmarks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 95, @@ -85,7 +85,7 @@ } ], "methodology": "Internal testing of output stability across effort levels and adaptive thinking", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Adaptive thinking removes the need to tune thinking budgets manually; the GA effort parameter (including new 'max') gives precise compute control" }, "latency_p50": { @@ -100,7 +100,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "5.5s", @@ -114,7 +114,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -128,7 +128,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -136,13 +136,13 @@ "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2026-06-01", - "value": "99.9%+ uptime (last 90 days)" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days)" } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Generational leap in abstract reasoning (68.8% ARC-AGI-2, ~2x Opus 4.5). 80.8% SWE-bench with 1M context and 128K output. Introduced adaptive thinking and GA effort parameter including 'max'. Now two generations behind Opus 4.8 but still fully served." @@ -163,7 +163,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 94, @@ -177,7 +177,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 87, @@ -191,7 +191,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 95, @@ -205,7 +205,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 88, @@ -219,7 +219,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong safety posture. Removal of last-assistant-turn prefills (400 error) eliminates a common response-manipulation pattern; structured outputs replace it." @@ -240,7 +240,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -254,7 +254,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -268,7 +268,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -282,7 +282,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -296,7 +296,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -310,7 +310,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy posture with ephemeral data handling and strong compliance certifications. HIPAA eligible for healthcare." @@ -331,7 +331,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 89, @@ -345,7 +345,7 @@ } ], "methodology": "Testing on factual QA datasets and real-world usage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -359,7 +359,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 88, @@ -373,7 +373,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -387,7 +387,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -401,7 +401,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 96, @@ -415,7 +415,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Adaptive thinking improves transparency by making reasoning depth model-driven and observable. Strong instruction following reduces need for aggressive prompt engineering." @@ -436,7 +436,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -450,7 +450,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -461,10 +461,16 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2026-02-05", "value": "Clear versioning with advance deprecation notice; Opus 4.6 remains served two generations behind Opus 4.8" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "Active; tentative retirement not sooner than February 5, 2027" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -478,7 +484,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -492,7 +498,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 93, @@ -506,7 +512,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -520,7 +526,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Mature operational profile with multi-cloud availability. Migration to 4.6 required removing assistant-turn prefills and moving to adaptive thinking — well-documented breaking changes." @@ -618,8 +624,8 @@ "pricing": { "input": "$5.00 per 1M tokens", "output": "$25.00 per 1M tokens", - "notes": "Same pricing as Opus 4.5. Batch API 50% discount. Prompt caching up to 90% savings.", - "last_verified": "2026-06-10" + "notes": "Same pricing as Opus 4.5. Batch API 50% discount. Prompt caching up to 90% savings. Confirmed unchanged at $5/$25 as of 2026-07-09.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 128000, @@ -641,7 +647,7 @@ "open_source": false, "architecture": "Transformer-based with Constitutional AI alignment, adaptive thinking, and effort parameter", "parameters": "Not disclosed", - "knowledge_cutoff": "Not disclosed" + "knowledge_cutoff": "May 2025 (reliable); training data through August 2025" }, "related_entities": ["claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-5", "claude-sonnet-4-6"], diff --git a/data/models/claude-opus-4-7.json b/data/models/claude-opus-4-7.json index 5ad1f89..d5fa2bc 100644 --- a/data/models/claude-opus-4-7.json +++ b/data/models/claude-opus-4-7.json @@ -4,7 +4,7 @@ "name": "Claude Opus 4.7", "provider": "Anthropic", "version": "20260416", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Previous-generation Opus flagship, superseded by Opus 4.8. 64.3% SWE-Bench Pro and 94.2% GPQA Diamond at launch. First Claude with high-resolution vision (2576px long edge, pixel-accurate coordinates), task budgets (beta), and the xhigh effort level.", "website": "https://www.anthropic.com/news/claude-opus-4-7", @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring real-world software engineering tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -44,7 +44,7 @@ } ], "methodology": "Graduate and PhD-level reasoning benchmarks evaluated with adaptive thinking at high effort", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 94, @@ -58,7 +58,7 @@ } ], "methodology": "Comprehensive knowledge and multimodal testing, including high-resolution screenshot and document understanding", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 93, @@ -72,7 +72,7 @@ } ], "methodology": "Repeated-run consistency testing across effort levels; structured extraction pipelines", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Adaptive thinking only; thinking content omitted by default; sampling parameters removed" }, "latency_p50": { @@ -87,7 +87,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "6.5s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -115,7 +115,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -123,13 +123,13 @@ "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2026-06-10", - "value": "99.9%+ uptime (last 90 days)" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days)" } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Was Anthropic's most capable model at launch (2026-04-16); now the previous-generation Opus behind Opus 4.8. Remains a strong, fully supported flagship-class choice, especially for vision-heavy workloads." @@ -149,7 +149,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 93, @@ -163,7 +163,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 88, @@ -177,7 +177,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 94, @@ -191,7 +191,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 89, @@ -205,7 +205,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Introduced real-time cybersecurity safeguards to the Opus line. Strong overall posture carried forward into Opus 4.8." @@ -225,7 +225,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -239,7 +239,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Configurable; zero-retention available", @@ -253,7 +253,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -267,7 +267,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -281,7 +281,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -295,7 +295,7 @@ } ], "methodology": "Review of data handling practices and trust center documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard Anthropic enterprise compliance posture: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API traffic." @@ -315,7 +315,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "First Opus where thinking text defaults to omitted \u2014 a transparency regression vs Opus 4.6 defaults, recoverable via display: summarized" }, "hallucination_rate": { @@ -330,7 +330,7 @@ } ], "methodology": "Testing on factual QA datasets and document-fidelity evaluations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -344,7 +344,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 87, @@ -358,7 +358,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -372,7 +372,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -386,7 +386,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 95, @@ -400,7 +400,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong documentation and guardrails. Thinking content omitted by default reduces out-of-the-box reasoning visibility; opt in to summarized display if reasoning is surfaced to users." @@ -420,7 +420,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Breaking changes vs Opus 4.6 (sampling params and manual thinking budgets return 400); Opus 4.8 keeps this same surface" }, "sdk_quality": { @@ -435,7 +435,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -446,10 +446,16 @@ "url": "https://platform.claude.com/docs/en/api/versioning", "date": "2026-04-16", "value": "Clear versioning with advance deprecation notice; claude-opus-4-7 alias remains active after Opus 4.8 launch" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "Active; tentative retirement not sooner than April 16, 2027" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -463,7 +469,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -477,7 +483,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 92, @@ -491,7 +497,7 @@ } ], "methodology": "Analysis of third-party integrations and availability surfaces", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -505,7 +511,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Mature operational profile. Superseded by Opus 4.8 as the flagship Opus, but remains fully supported at the same $5/$25 price; upgrade to 4.8 is a drop-in model-ID swap." @@ -624,8 +630,8 @@ "pricing": { "input": "$5.00 per 1M tokens", "output": "$25.00 per 1M tokens", - "notes": "1M context at standard API pricing with no long-context premium. Batch API 50% discount, prompt caching savings apply. No fast-mode variant.", - "last_verified": "2026-06-10" + "notes": "1M context at standard API pricing with no long-context premium. Batch API 50% discount, prompt caching savings apply. No fast-mode variant. Confirmed unchanged at $5/$25 as of 2026-07-09.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 128000, @@ -653,7 +659,7 @@ "open_source": false, "architecture": "Transformer-based with Constitutional AI alignment; adaptive thinking only with effort parameter introducing xhigh; high-resolution vision", "parameters": "Not disclosed", - "knowledge_cutoff": "Not disclosed", + "knowledge_cutoff": "January 2026 (reliable knowledge and training data cutoff)", "release_date": "2026-04-16" }, "related_entities": [ diff --git a/data/models/claude-opus-4-8.json b/data/models/claude-opus-4-8.json index 2e7c73f..4b07b62 100644 --- a/data/models/claude-opus-4-8.json +++ b/data/models/claude-opus-4-8.json @@ -4,7 +4,7 @@ "name": "Claude Opus 4.8", "provider": "Anthropic", "version": "20260528", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Anthropic's flagship Opus model with state-of-the-art long-horizon agentic execution, knowledge work, and memory. 84% on Online-Mind2Web, dynamic multi-subagent workflows, ~4x less likely to miss its own code flaws than its predecessor, and 1M context at standard pricing.", "website": "https://www.anthropic.com/news/claude-opus-4-8", @@ -30,7 +30,7 @@ } ], "methodology": "Agentic coding and web-agent benchmarks measuring long-horizon autonomous execution and self-verification", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -44,7 +44,7 @@ } ], "methodology": "Graduate-level reasoning and knowledge-work benchmarks evaluated with adaptive thinking at high effort", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 95, @@ -58,7 +58,7 @@ } ], "methodology": "Comprehensive knowledge and multimodal testing, including high-resolution vision inherited from Opus 4.7", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 95, @@ -72,7 +72,7 @@ } ], "methodology": "Long-horizon agentic run consistency and self-verification testing across effort levels", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Same adaptive-thinking-only surface as Opus 4.7; effort parameter (incl. xhigh) controls depth" }, "latency_p50": { @@ -87,7 +87,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "6.5s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -115,21 +115,22 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { - "score": 99, + "score": 98, "confidence": "high", "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2026-06-10", - "value": "99.9%+ uptime (last 90 days)" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days); elevated-error incidents across models in early July 2026, including an Opus 4.8-specific incident on 2026-07-09" } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09", + "notes": "Score trimmed from 99 to 98: 90-day API uptime dipped below 99.9% amid early-July 2026 incident cluster" } }, "notes": "Current flagship Opus. State-of-the-art long-horizon agentic execution, knowledge work, and memory; 84% Online-Mind2Web; dynamic multi-subagent workflows. Superseded only by the higher-tier Claude Fable 5." @@ -149,7 +150,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks, including agentic browsing scenarios", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 94, @@ -163,7 +164,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 88, @@ -177,7 +178,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 95, @@ -191,7 +192,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 89, @@ -205,7 +206,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong safety posture with agentic-specific safeguards. Mid-session system prompts (beta) give operators a non-spoofable instruction channel for long-running sessions." @@ -225,7 +226,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -239,7 +240,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Configurable; zero-retention available", @@ -253,7 +254,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -267,7 +268,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -281,7 +282,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -295,7 +296,7 @@ } ], "methodology": "Review of data handling practices and trust center documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard Anthropic enterprise compliance posture: SOC 2 Type II, GDPR, HIPAA-eligible, training opt-out by default for API traffic." @@ -315,7 +316,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 89, @@ -329,7 +330,7 @@ } ], "methodology": "Testing on factual QA datasets and self-verification evaluations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -343,7 +344,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 88, @@ -357,7 +358,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 93, @@ -371,7 +372,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -385,7 +386,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 96, @@ -399,7 +400,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "More deliberate and transparent in long agentic runs than Opus 4.7 \u2014 narrates progress, flags uncertainty, and self-verifies code. Thinking text remains omitted by default." @@ -419,7 +420,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Adaptive thinking only; effort levels include xhigh; task budgets (beta) supported" }, "sdk_quality": { @@ -434,7 +435,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -445,10 +446,16 @@ "url": "https://platform.claude.com/docs/en/api/versioning", "date": "2026-05-28", "value": "Clear versioning with advance deprecation notice; stable claude-opus-4-8 alias" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "Active; tentative retirement not sooner than May 28, 2027; recommended replacement for deprecated Opus 4.1 and retired Opus 4" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -462,7 +469,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -476,7 +483,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 92, @@ -490,7 +497,7 @@ } ], "methodology": "Analysis of third-party integrations and availability surfaces", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -504,7 +511,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Drop-in upgrade from Opus 4.7 (identical API surface). 1M context at standard pricing with no long-context premium; optional fast mode at $10/$50." @@ -625,8 +632,8 @@ "pricing": { "input": "$5.00 per 1M tokens", "output": "$25.00 per 1M tokens", - "notes": "Fast mode available at $10/$50 per 1M. 1M context at standard pricing with no long-context premium. Batch API 50% discount, prompt caching savings apply.", - "last_verified": "2026-06-10" + "notes": "Fast mode available at $10/$50 per 1M. 1M context at standard pricing with no long-context premium. Batch API 50% discount, prompt caching savings apply. Confirmed unchanged at $5/$25 as of 2026-07-09.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 128000, @@ -654,7 +661,7 @@ "open_source": false, "architecture": "Transformer-based with Constitutional AI alignment; adaptive thinking only with effort parameter including xhigh; same API surface as Opus 4.7", "parameters": "Not disclosed", - "knowledge_cutoff": "Not disclosed", + "knowledge_cutoff": "January 2026 (reliable knowledge and training data cutoff)", "release_date": "2026-05-28" }, "related_entities": [ diff --git a/data/models/claude-opus-4.json b/data/models/claude-opus-4.json index d5ca1a7..849c2fb 100644 --- a/data/models/claude-opus-4.json +++ b/data/models/claude-opus-4.json @@ -3,10 +3,10 @@ "type": "model", "name": "Claude Opus 4", "provider": "Anthropic", - "version": "20250522", - "last_evaluated": "2025-11-17", + "version": "20250514", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Anthropic's most powerful model released May 2025. Exceptional reasoning, coding (72.5-79.4% SWE-bench in high-compute), and agentic capabilities.", + "description": "RETIRED: Anthropic retired Claude Opus 4 (claude-opus-4-20250514) on 2026-06-15 (deprecated 2026-04-14); API requests now fail. Recommended replacement: Claude Opus 4.8. Historically Anthropic's most powerful model of May 2025, with exceptional reasoning, coding (72.5-79.4% SWE-bench in high-compute), and agentic capabilities.", "website": "https://www.anthropic.com/claude", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -45,7 +45,7 @@ } ], "methodology": "PhD-level reasoning benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 94, @@ -65,7 +65,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 92, @@ -79,7 +79,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "2.5s", @@ -93,7 +93,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "5.5s", @@ -107,7 +107,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "context_window": { "value": "200,000 tokens", @@ -121,7 +121,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -135,10 +135,10 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, - "notes": "Best-in-class performance. 79.4% SWE-bench in high-compute mode (highest). 90% AIME in high-compute. Exceptional for complex reasoning." + "notes": "Historical evaluation: best-in-class performance at release (79.4% SWE-bench high-compute, 90% AIME high-compute). Model retired 2026-06-15 and is no longer served." }, "security": { @@ -156,7 +156,7 @@ } ], "methodology": "OWASP LLM01 testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 93, @@ -170,7 +170,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 89, @@ -184,7 +184,7 @@ } ], "methodology": "Privacy policy analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_safety": { "score": 94, @@ -198,7 +198,7 @@ } ], "methodology": "Comprehensive safety testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "api_security": { "score": 90, @@ -212,7 +212,7 @@ } ], "methodology": "Security features review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Flagship security with ASL-3 standard and Constitutional AI. Strongest safety guardrails." @@ -233,7 +233,7 @@ } ], "methodology": "Enterprise documentation review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 97, @@ -247,7 +247,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -261,7 +261,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 90, @@ -275,7 +275,7 @@ } ], "methodology": "Data protection review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -289,7 +289,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 97, @@ -303,7 +303,7 @@ } ], "methodology": "Data handling review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy. Ephemeral data handling, HIPAA eligible, strongest compliance for regulated industries." @@ -324,7 +324,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 87, @@ -338,7 +338,7 @@ } ], "methodology": "Factual QA testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 86, @@ -352,7 +352,7 @@ } ], "methodology": "Bias benchmark evaluation", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 88, @@ -366,7 +366,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 93, @@ -380,7 +380,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 81, @@ -394,7 +394,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "guardrails": { "score": 95, @@ -408,14 +408,14 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Excellent transparency with extended thinking and comprehensive system card. Best-in-class guardrails." }, "operational_excellence": { - "overall_score": 92, + "overall_score": 78, "criteria": { "api_design_quality": { "score": 94, @@ -429,7 +429,7 @@ } ], "methodology": "API design review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -443,10 +443,10 @@ } ], "methodology": "SDK quality review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 90, + "score": 65, "confidence": "high", "evidence": [ { @@ -454,10 +454,17 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2025-05-22", "value": "6-month deprecation notice" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "claude-opus-4-20250514 deprecated 2026-04-14 and retired 2026-06-15; requests fail; recommended replacement claude-opus-4-8" } ], "methodology": "Versioning policy review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09", + "notes": "Score reduced from 90: model retired 2026-06-15 and no longer available on the API" }, "monitoring_observability": { "score": 89, @@ -471,7 +478,7 @@ } ], "methodology": "Monitoring tools review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "support_quality": { "score": 92, @@ -485,10 +492,10 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { - "score": 92, + "score": 72, "confidence": "high", "evidence": [ { @@ -499,7 +506,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "license_terms": { "score": 93, @@ -513,63 +520,63 @@ } ], "methodology": "License review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, - "notes": "Flagship operational excellence. Available on API, Amazon Bedrock, and Google Vertex AI." + "notes": "Model retired 2026-06-15 on Anthropic-operated platforms; API requests fail. Migration target is Claude Opus 4.8. Versioning, ecosystem, and overall scores reduced to reflect retirement." } }, "use_case_ratings": { "code-generation": { "overall": 97, - "notes": "Best-in-class coding. 79.4% SWE-bench in high-compute mode. Exceptional for complex software engineering.", - "alternatives": ["claude-sonnet-4-5", "claude-haiku-4-5"] + "notes": "Historically best-in-class coding (79.4% SWE-bench high-compute). Retired — use Opus 4.8.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "customer-support": { "overall": 90, "notes": "Excellent but potentially over-powered and expensive for standard customer support.", - "alternatives": ["claude-sonnet-4", "claude-haiku-4-5"] + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] }, "content-creation": { "overall": 94, "notes": "Exceptional creative writing with nuanced understanding and natural style.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] }, "data-analysis": { "overall": 93, "notes": "Superior analytical capabilities with extended thinking for complex analysis.", - "alternatives": ["openai-o3", "claude-sonnet-4"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "research-assistant": { "overall": 95, "notes": "Outstanding for research. Extended thinking enables deep analysis. 200K context for long documents.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-opus-4-8", "gemini-3-1-pro"] }, "legal-compliance": { "overall": 95, - "notes": "Best for legal work. HIPAA eligible, ephemeral data, ASL-3 security. Careful reasoning.", - "alternatives": ["claude-opus-4-1"] + "notes": "Best for legal work in its era. HIPAA eligible, ephemeral data, ASL-3 security. Careful reasoning.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "healthcare": { "overall": 94, - "notes": "Flagship for healthcare. HIPAA eligible, strongest privacy, careful medical reasoning.", - "alternatives": ["claude-opus-4-1", "claude-sonnet-4"] + "notes": "Flagship for healthcare in its era. HIPAA eligible, strongest privacy, careful medical reasoning.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "financial-analysis": { "overall": 92, "notes": "Exceptional for complex financial modeling and analysis. 90% AIME math in high-compute.", - "alternatives": ["openai-o3", "claude-opus-4-1"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] }, "education": { "overall": 92, "notes": "Excellent for education with patient, detailed explanations and strong knowledge base.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] }, "creative-writing": { "overall": 93, "notes": "Outstanding creative capabilities with nuanced character development and storytelling.", - "alternatives": ["gpt-5", "claude-sonnet-4-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] } }, @@ -583,6 +590,7 @@ ], "limitations": [ + "RETIRED 2026-06-15 — no longer available on the Claude API; requests fail (migrate to Claude Opus 4.8)", "Premium pricing ($15/$75 per 1M tokens)", "Higher latency (~2.5s p50, 5.5s p95)", "Training cutoff March 2025", @@ -599,6 +607,7 @@ ], "not_recommended_for": [ + "Any use — the model is retired and API requests fail; migrate to Claude Opus 4.8", "Latency-sensitive real-time applications", "High-volume cost-sensitive workloads", "Simple tasks better handled by smaller models", @@ -609,8 +618,8 @@ "pricing": { "input": "$15.00 per 1M tokens", "output": "$75.00 per 1M tokens", - "notes": "Flagship pricing. Batch API 50% discount. Prompt caching up to 90% savings.", - "last_verified": "2025-11-17" + "notes": "Historical flagship pricing. Model retired 2026-06-15 — no longer purchasable on Anthropic-operated platforms.", + "last_verified": "2026-07-09" }, "context_window": 200000, "max_output_tokens": 32000, @@ -636,10 +645,10 @@ "safety_level": "ASL-3" }, - "related_entities": ["claude-opus-4-1", "claude-sonnet-4-5", "claude-sonnet-4", "gpt-5"], + "related_entities": ["claude-opus-4-8", "claude-opus-4-1", "claude-sonnet-4-6", "claude-haiku-4-5"], "tags": [ - "flagship", + "retired", "coding", "reasoning", "hipaa-eligible", diff --git a/data/models/claude-sonnet-4-5.json b/data/models/claude-sonnet-4-5.json index e925290..a10c999 100644 --- a/data/models/claude-sonnet-4-5.json +++ b/data/models/claude-sonnet-4-5.json @@ -4,9 +4,9 @@ "name": "Claude Sonnet 4.5", "provider": "Anthropic", "version": "20250929", - "last_evaluated": "2025-11-07", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "State-of-the-art AI model with exceptional coding capabilities, extended thinking, and strong safety features. Best-in-class for software development tasks.", + "description": "Previous-generation Sonnet released September 2025, since superseded by Sonnet 4.6 (2026) and Claude Sonnet 5 (2026-06-30). Still Active on the API with tentative retirement not sooner than 2026-09-29. Historically the top coding model of its era (77.2% SWE-bench Verified at launch) with extended thinking and strong safety features.", "website": "https://www.anthropic.com/claude", "trust_vector": { @@ -19,9 +19,9 @@ "evidence": [ { "source": "SWE-bench Verified", - "url": "https://www.anthropic.com/news/claude-3-5-sonnet", - "date": "2024-10-22", - "value": "49.0% resolution rate (highest on benchmark)" + "url": "https://www.anthropic.com/news/claude-sonnet-4-5", + "date": "2025-09-29", + "value": "77.2% resolution rate (highest of any model at launch); corrected 2026-07-09 from a mislabeled Claude 3.5 Sonnet figure" }, { "source": "Anthropic Internal Benchmarks", @@ -31,7 +31,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring real-world software engineering tasks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 92, @@ -51,17 +51,17 @@ } ], "methodology": "Graduate and PhD-level reasoning benchmarks requiring multi-step problem solving", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 93, "confidence": "high", "evidence": [ { - "source": "LMSYS Chatbot Arena", - "url": "https://lmsys.org/blog/2024-10-17-arena-update/", - "date": "2024-10-17", - "value": "1324 ELO (Rank #2 overall)" + "source": "OSWorld", + "url": "https://www.anthropic.com/news/claude-sonnet-4-5", + "date": "2025-09-29", + "value": "61.4% on real-world computer-use tasks (benchmark leader at launch)" }, { "source": "MMLU-Pro", @@ -71,7 +71,7 @@ } ], "methodology": "Crowdsourced blind comparisons and comprehensive knowledge testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 91, @@ -85,7 +85,7 @@ } ], "methodology": "Internal testing with repeated prompts at various temperature settings", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Extended thinking feature provides more consistent reasoning paths" }, "latency_p50": { @@ -100,7 +100,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.2s", @@ -114,7 +114,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "context_window": { "value": "200,000 tokens", @@ -128,7 +128,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -136,13 +136,13 @@ "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2025-11-01", - "value": "99.95% uptime (last 90 days)" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days)" } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Exceptional performance across coding, reasoning, and general tasks. Extended thinking capability enables more reliable outputs for complex problems." @@ -169,7 +169,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 92, @@ -189,7 +189,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -203,7 +203,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Strong policies, but inherent LLM memorization risks exist" }, "output_safety": { @@ -218,7 +218,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "api_security": { "score": 88, @@ -232,7 +232,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Strong security posture with Constitutional AI providing robust guardrails. Best-in-class prompt injection resistance." @@ -253,7 +253,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -267,7 +267,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -281,7 +281,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -295,7 +295,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "No built-in PII detection, customers must implement their own controls" }, "compliance_certifications": { @@ -310,7 +310,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -324,7 +324,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy posture with ephemeral data handling and strong compliance certifications. HIPAA eligible." @@ -351,7 +351,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 86, @@ -371,7 +371,7 @@ } ], "methodology": "Testing on factual QA datasets and real-world usage", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Extended thinking mode reduces hallucinations further" }, "bias_fairness": { @@ -392,7 +392,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Ongoing work, Constitutional AI helps but not perfect" }, "uncertainty_quantification": { @@ -407,7 +407,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "No explicit confidence scores, relies on natural language expression" }, "model_card_quality": { @@ -422,7 +422,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -436,7 +436,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Limited transparency on specific training data sources (industry standard)" }, "guardrails": { @@ -451,7 +451,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Strong explainability with extended thinking feature. Constitutional AI provides transparency in alignment approach. Training data transparency could be improved." @@ -472,7 +472,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -486,7 +486,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -497,10 +497,16 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2025-10-01", "value": "Clear versioning with 6-month deprecation notice" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "claude-sonnet-4-5-20250929 still Active (not deprecated); tentative retirement not sooner than September 29, 2026; listed as a legacy model in the docs" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -514,7 +520,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Basic metrics available, but no detailed request tracing" }, "support_quality": { @@ -529,7 +535,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 91, @@ -543,7 +549,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -557,7 +563,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Excellent operational maturity with well-designed APIs, strong SDKs, and good documentation. Enterprise-ready." @@ -567,58 +573,58 @@ "use_case_ratings": { "code-generation": { "overall": 96, - "notes": "Best-in-class for code generation. Exceptional at Python, TypeScript, and explaining code. Extended thinking helps with complex architectural decisions.", - "alternatives": ["gpt-5", "claude-opus-4-1"] + "notes": "Top coding model of its era (77.2% SWE-bench Verified at launch). Superseded by Sonnet 4.6 and Sonnet 5 for new builds.", + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "customer-support": { "overall": 88, "notes": "Strong empathy and natural conversation. Slightly higher latency than specialized models, but excellent quality.", - "alternatives": ["gpt-5"] + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] }, "content-creation": { "overall": 90, "notes": "Excellent for long-form content, maintains consistent voice and structure. Natural writing style.", - "alternatives": ["gpt-5", "claude-opus-4-1"] + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "data-analysis": { "overall": 93, "notes": "Strong SQL generation and data interpretation. Extended thinking excellent for complex analytical tasks.", - "alternatives": ["claude-opus-4-1"] + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "research-assistant": { "overall": 91, "notes": "Excellent summarization and synthesis. Extended thinking mode provides detailed reasoning for complex topics.", - "alternatives": ["claude-opus-4-1", "gpt-5"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "legal-compliance": { "overall": 89, "notes": "Strong privacy posture and careful reasoning. HIPAA eligible. Extended thinking useful for contract analysis.", - "alternatives": ["claude-opus-4-1"] + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "healthcare": { "overall": 87, "notes": "HIPAA eligible with strong privacy controls. Good for clinical documentation but requires human oversight.", - "alternatives": ["claude-opus-4-1"] + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "financial-analysis": { "overall": 90, "notes": "Strong analytical capabilities and mathematical reasoning. Good for financial modeling and report generation.", - "alternatives": ["claude-opus-4-1", "gpt-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] }, "education": { "overall": 92, "notes": "Excellent tutoring capabilities with patient explanations. Extended thinking shows work step-by-step.", - "alternatives": ["gpt-5", "claude-opus-4-1"] + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] }, "creative-writing": { "overall": 88, "notes": "Good for creative tasks but can be slightly verbose. Strong dialogue and character development.", - "alternatives": ["claude-opus-4-1", "gpt-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] } }, "strengths": [ - "Best-in-class coding capabilities (SWE-bench leader)", + "Top coding model of its era: 77.2% SWE-bench Verified at launch (September 2025)", "Extended thinking feature for complex problem-solving", "Exceptional privacy posture with ephemeral data handling", "Strong safety and jailbreak resistance via Constitutional AI", @@ -631,7 +637,9 @@ "Limited vision capabilities compared to multimodal specialists", "Training data transparency could be improved", "No built-in PII detection (customer responsibility)", - "Premium pricing ($3/$15 per 1M tokens)" + "Premium pricing ($3/$15 per 1M tokens)", + "Superseded by Sonnet 4.6 and Claude Sonnet 5 (2026-06-30) at the same standard price; 200K context vs their 1M", + "No adaptive thinking or effort parameter (introduced with Sonnet 4.6)" ], "best_for": [ @@ -643,6 +651,7 @@ ], "not_recommended_for": [ + "New Sonnet-tier deployments — Sonnet 4.6 or Claude Sonnet 5 offer more capability at the same standard price", "Real-time applications requiring <500ms latency", "Vision-heavy applications (limited multimodal)", "Cost-sensitive projects needing high volume inference" @@ -652,9 +661,11 @@ "pricing": { "input": "$3.00 per 1M tokens", "output": "$15.00 per 1M tokens", - "notes": "Premium tier pricing, batch discounts available for enterprise" + "notes": "Premium tier pricing, batch discounts available for enterprise. Confirmed unchanged at $3/$15 as of 2026-07-09.", + "last_verified": "2026-07-09" }, "context_window": 200000, + "max_output": 64000, "languages": [ "English", "Spanish", @@ -672,10 +683,11 @@ "api_endpoint": "https://api.anthropic.com/v1/messages", "open_source": false, "architecture": "Transformer-based with Constitutional AI alignment", - "parameters": "Not disclosed" + "parameters": "Not disclosed", + "knowledge_cutoff": "January 2025 (reliable); training data through July 2025" }, - "related_entities": ["claude-opus-4-1", "gpt-5", "gemini-2-5-pro"], + "related_entities": ["claude-sonnet-4-6", "claude-opus-4-8", "claude-haiku-4-5", "gpt-5-5"], "tags": [ "coding", @@ -683,6 +695,7 @@ "enterprise", "hipaa-eligible", "safety-focused", - "extended-thinking" + "extended-thinking", + "previous-generation" ] } diff --git a/data/models/claude-sonnet-4-6.json b/data/models/claude-sonnet-4-6.json index 5a5fbf5..8fe85ec 100644 --- a/data/models/claude-sonnet-4-6.json +++ b/data/models/claude-sonnet-4-6.json @@ -4,9 +4,9 @@ "name": "Claude Sonnet 4.6", "provider": "Anthropic", "version": "4.6", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Anthropic's best speed/intelligence balance — the value workhorse for agentic and production workloads at $3/$15 per 1M tokens, with a 1M token context window, adaptive thinking, the effort parameter including 'max', and strong computer-use accuracy.", + "description": "Anthropic's previous-generation Sonnet workhorse, superseded by Claude Sonnet 5 (2026-06-30) as the best speed/intelligence balance at the same $3/$15 price. Still fully supported (Active, tentative retirement not sooner than 2027-02-17), with a 1M token context window, 128K max output, adaptive thinking, the effort parameter including 'max', and strong computer-use accuracy.", "website": "https://www.anthropic.com/claude/sonnet", "trust_vector": { @@ -23,6 +23,12 @@ "date": "2026-06-10", "value": "Recommended model for agentic coding at the Sonnet tier; successor to Claude Sonnet 4.5" }, + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Sonnet 4.6 scores 62.3% SWE-bench Verified and 55.4% Terminal-bench vs Sonnet 5's 72.7% and 76.1%; Sonnet 5 is now the recommended Sonnet-tier model" + }, { "source": "Anthropic Migration Guide", "url": "https://platform.claude.com/docs/en/about-claude/models/overview", @@ -31,7 +37,7 @@ } ], "methodology": "Review of official model documentation and positioning for software engineering workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 90, @@ -45,7 +51,7 @@ } ], "methodology": "Review of documented thinking capabilities and reasoning benchmark positioning", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 92, @@ -59,7 +65,7 @@ } ], "methodology": "Comprehensive knowledge and multimodal capability review against official documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 93, @@ -73,7 +79,7 @@ } ], "methodology": "Internal testing of output stability across effort levels and adaptive thinking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.5s", @@ -87,7 +93,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.5s", @@ -101,7 +107,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -110,12 +116,12 @@ { "source": "Anthropic Models Documentation", "url": "https://platform.claude.com/docs/en/about-claude/models/overview", - "date": "2026-06-10", - "value": "1M token context window; 64K max output tokens" + "date": "2026-07-09", + "value": "1M token context window; 128K max output tokens (up to 300K on the Message Batches API via the extended-output beta)" } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -123,16 +129,16 @@ "evidence": [ { "source": "Anthropic Status Page", - "url": "https://status.anthropic.com/", - "date": "2026-06-01", - "value": "99.9%+ uptime (last 90 days)" + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days)" } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "The value workhorse of the Claude lineup: near-Opus intelligence at Sonnet latency and price, with a 1M context window, adaptive thinking, and the full effort range including 'max'." + "notes": "Former value workhorse of the Claude lineup: near-Opus intelligence at Sonnet latency and price, with a 1M context window, adaptive thinking, and the full effort range including 'max'. Superseded by Claude Sonnet 5 (2026-06-30, same $3/$15 with intro $2/$10 through 2026-08-31) — prefer Sonnet 5 for new builds." }, "security": { @@ -150,7 +156,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 93, @@ -164,7 +170,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 87, @@ -178,7 +184,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 94, @@ -192,7 +198,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 88, @@ -206,7 +212,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong safety posture. Like Opus 4.6, last-assistant-turn prefills return a 400 — structured outputs (output_config.format) are the supported replacement." @@ -227,7 +233,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -241,7 +247,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -255,7 +261,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 88, @@ -269,7 +275,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -283,7 +289,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -297,7 +303,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Same enterprise-grade privacy posture as the Opus tier: ephemeral data handling, strong certifications, HIPAA eligible." @@ -318,7 +324,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 87, @@ -332,7 +338,7 @@ } ], "methodology": "Testing on factual QA datasets and real-world usage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -346,7 +352,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -360,7 +366,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 91, @@ -374,7 +380,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -388,7 +394,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 95, @@ -402,7 +408,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Transparent compute controls (adaptive thinking + effort) and thorough migration documentation. Follows instructions closely, reducing prompt-engineering opacity." @@ -423,7 +429,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -437,7 +443,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -448,10 +454,16 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2026-06-10", "value": "Clear versioning with advance deprecation notice; documented migration path from Sonnet 4.5 and retired 3.x Sonnets" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "Active; tentative retirement not sooner than February 17, 2027; recommended replacement for retired Sonnet 4" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -465,7 +477,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -479,7 +491,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 93, @@ -493,7 +505,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -507,7 +519,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Production-ready with multi-cloud availability. Migration from Sonnet 4.5 requires setting effort explicitly (4.6 defaults to high) and removing assistant prefills." @@ -568,8 +580,8 @@ }, "strengths": [ - "Best speed/intelligence balance in the Claude lineup at $3/$15 per 1M tokens", - "1M token context window with 64K max output", + "Strong speed/intelligence balance at $3/$15 per 1M tokens (Sonnet 5 now leads the tier)", + "1M token context window with 128K max output (300K on the Batch API extended-output beta)", "Adaptive thinking supported — no manual thinking budgets to tune", "Effort parameter including 'max' (not available on Sonnet 4.5 or Haiku)", "Strong computer-use accuracy for agentic automation", @@ -581,7 +593,7 @@ "Lower ceiling than Opus tier on the hardest reasoning and long-horizon agentic tasks", "Removed assistant prefills — code relying on prefills returns 400", "Effort defaults to high — Sonnet 4.5 migrations see higher latency/cost unless effort is set explicitly", - "64K max output (vs 128K on Opus 4.6+)", + "Superseded by Claude Sonnet 5 (2026-06-30): 62.3% vs 72.7% SWE-bench Verified at the same standard price", "No native audio capabilities" ], @@ -595,8 +607,8 @@ "not_recommended_for": [ "Frontier-difficulty reasoning where Opus 4.8 is warranted", + "New Sonnet-tier deployments — Claude Sonnet 5 offers materially better agentic performance at the same standard price", "Workflows still dependent on assistant-turn prefills", - "Outputs beyond 64K tokens in a single response", "Audio processing applications" ], @@ -604,11 +616,11 @@ "pricing": { "input": "$3.00 per 1M tokens", "output": "$15.00 per 1M tokens", - "notes": "Same pricing as Sonnet 4.5. Batch API 50% discount. Prompt caching up to 90% savings.", - "last_verified": "2026-06-10" + "notes": "Same pricing as Sonnet 4.5. Batch API 50% discount. Prompt caching up to 90% savings. Confirmed unchanged at $3/$15 as of 2026-07-09; successor Sonnet 5 has the same standard price (intro $2/$10 through 2026-08-31).", + "last_verified": "2026-07-09" }, "context_window": 1000000, - "max_output": 64000, + "max_output": 128000, "languages": [ "English", "Spanish", @@ -627,7 +639,7 @@ "open_source": false, "architecture": "Transformer-based with Constitutional AI alignment, adaptive thinking, and effort parameter", "parameters": "Not disclosed", - "knowledge_cutoff": "Not disclosed" + "knowledge_cutoff": "August 2025 (reliable); training data through January 2026" }, "related_entities": ["claude-sonnet-4-5", "claude-opus-4-6", "claude-opus-4-8", "claude-haiku-4-5"], @@ -642,6 +654,7 @@ "effort-parameter", "computer-use", "long-context", - "value-workhorse" + "value-workhorse", + "previous-generation" ] } diff --git a/data/models/claude-sonnet-4.json b/data/models/claude-sonnet-4.json index 4b31f36..d9710a0 100644 --- a/data/models/claude-sonnet-4.json +++ b/data/models/claude-sonnet-4.json @@ -3,10 +3,10 @@ "type": "model", "name": "Claude Sonnet 4", "provider": "Anthropic", - "version": "20250522", - "last_evaluated": "2025-11-17", + "version": "20250514", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Anthropic's Claude Sonnet 4 model released May 2025 with exceptional coding capabilities and advanced reasoning. Hybrid model with extended thinking mode.", + "description": "RETIRED: Anthropic retired Claude Sonnet 4 (claude-sonnet-4-20250514) on 2026-06-15 (deprecated 2026-04-14); API requests now fail. Recommended replacement: Claude Sonnet 4.6. Historically a May 2025 hybrid model with exceptional coding capabilities, advanced reasoning, and extended thinking mode.", "website": "https://www.anthropic.com/claude", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 91, @@ -39,7 +39,7 @@ } ], "methodology": "Graduate-level reasoning benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 90, @@ -53,7 +53,7 @@ } ], "methodology": "Knowledge testing benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 89, @@ -67,7 +67,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.5s", @@ -81,7 +81,7 @@ } ], "methodology": "Median latency for API requests", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.2s", @@ -95,7 +95,7 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "context_window": { "value": "200,000 tokens", @@ -109,7 +109,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -123,10 +123,10 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, - "notes": "Exceptional coding performance with 72.7% SWE-bench. Hybrid model with extended thinking for complex tasks. 200K context window for large codebases." + "notes": "Historical evaluation: exceptional coding performance at release (72.7% SWE-bench) with extended thinking and 200K context. Model retired 2026-06-15 and is no longer served." }, "security": { @@ -144,7 +144,7 @@ } ], "methodology": "Testing against OWASP LLM01 attacks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 92, @@ -158,7 +158,7 @@ } ], "methodology": "Testing against adversarial prompts", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 88, @@ -172,7 +172,7 @@ } ], "methodology": "Analysis of privacy policies", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_safety": { "score": 93, @@ -186,7 +186,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "api_security": { "score": 89, @@ -200,7 +200,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Excellent security with Constitutional AI providing strong guardrails. Best-in-class safety for enterprise use." @@ -221,7 +221,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 96, @@ -235,7 +235,7 @@ } ], "methodology": "Analysis of privacy policy", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -249,7 +249,7 @@ } ], "methodology": "Review of terms of service", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 89, @@ -263,7 +263,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 93, @@ -277,7 +277,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 96, @@ -291,7 +291,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy with ephemeral data handling. HIPAA eligible. Strong compliance posture for regulated industries." @@ -312,7 +312,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 85, @@ -326,7 +326,7 @@ } ], "methodology": "Testing on factual QA datasets", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -340,7 +340,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -354,7 +354,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 91, @@ -368,7 +368,7 @@ } ], "methodology": "Review of documentation", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 79, @@ -382,7 +382,7 @@ } ], "methodology": "Review of public disclosures", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "guardrails": { "score": 94, @@ -396,14 +396,14 @@ } ], "methodology": "Analysis of safety mechanisms", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency with Constitutional AI and extended thinking feature. Comprehensive model card available." }, "operational_excellence": { - "overall_score": 91, + "overall_score": 78, "criteria": { "api_design_quality": { "score": 93, @@ -417,7 +417,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -431,10 +431,10 @@ } ], "methodology": "Review of SDK quality", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 89, + "score": 65, "confidence": "high", "evidence": [ { @@ -442,10 +442,17 @@ "url": "https://docs.anthropic.com/en/api/versioning", "date": "2025-05-22", "value": "6-month deprecation notice" + }, + { + "source": "Anthropic Model Deprecations", + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "date": "2026-07-09", + "value": "claude-sonnet-4-20250514 deprecated 2026-04-14 and retired 2026-06-15; requests fail; recommended replacement claude-sonnet-4-6" } ], "methodology": "Review of versioning", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09", + "notes": "Score reduced from 89: model retired 2026-06-15 and no longer available on the API" }, "monitoring_observability": { "score": 88, @@ -459,7 +466,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "support_quality": { "score": 90, @@ -473,10 +480,10 @@ } ], "methodology": "Assessment of support", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { - "score": 91, + "score": 72, "confidence": "high", "evidence": [ { @@ -487,7 +494,7 @@ } ], "methodology": "Analysis of ecosystem", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -501,63 +508,63 @@ } ], "methodology": "Review of licensing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, - "notes": "Excellent operational maturity with well-designed APIs and strong developer experience. Available on API, Bedrock, and Vertex AI." + "notes": "Model retired 2026-06-15 on Anthropic-operated platforms; API requests fail. Migration target is Claude Sonnet 4.6. Versioning, ecosystem, and overall scores reduced to reflect retirement." } }, "use_case_ratings": { "code-generation": { "overall": 95, - "notes": "Exceptional coding capabilities with 72.7% SWE-bench. Best for complex software engineering tasks.", - "alternatives": ["claude-sonnet-4-5", "claude-opus-4"] + "notes": "Historically exceptional coding (72.7% SWE-bench). Retired — use Sonnet 4.6 or newer.", + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "customer-support": { "overall": 89, "notes": "Excellent for customer support with empathetic responses and fast latency.", - "alternatives": ["claude-sonnet-4-5", "gpt-5"] + "alternatives": ["claude-sonnet-4-6", "claude-haiku-4-5"] }, "content-creation": { "overall": 88, "notes": "Strong content creation with natural writing style. Large context window helpful.", - "alternatives": ["claude-opus-4", "gpt-5"] + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "data-analysis": { "overall": 87, "notes": "Strong analytical capabilities with extended thinking for complex analysis.", - "alternatives": ["claude-opus-4", "openai-o3"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "research-assistant": { "overall": 90, "notes": "Excellent for research with strong summarization and 200K context window.", - "alternatives": ["claude-opus-4", "gpt-5"] + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] }, "legal-compliance": { "overall": 91, "notes": "Strong privacy posture (HIPAA eligible) and careful reasoning for legal tasks.", - "alternatives": ["claude-opus-4"] + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "healthcare": { "overall": 90, "notes": "HIPAA eligible with strong privacy. Good for clinical documentation with oversight.", - "alternatives": ["claude-opus-4"] + "alternatives": ["claude-sonnet-4-6", "claude-opus-4-8"] }, "financial-analysis": { "overall": 88, "notes": "Strong financial analysis with extended thinking for complex modeling.", - "alternatives": ["claude-opus-4", "openai-o3"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] }, "education": { "overall": 89, "notes": "Good for educational content with patient explanations and strong knowledge.", - "alternatives": ["claude-sonnet-4-5", "gpt-5"] + "alternatives": ["claude-sonnet-4-6", "gpt-5-5"] }, "creative-writing": { "overall": 87, "notes": "Strong creative writing with natural style. Good dialogue and character development.", - "alternatives": ["claude-opus-4", "gpt-5"] + "alternatives": ["claude-opus-4-8", "gpt-5-5"] } }, @@ -571,11 +578,12 @@ ], "limitations": [ + "RETIRED 2026-06-15 — no longer available on the Claude API; requests fail (migrate to Claude Sonnet 4.6)", "Higher latency in extended thinking mode", "Training data cutoff March 2025", "No built-in PII detection", "Premium pricing ($3/$15 per 1M tokens)", - "Superseded by Sonnet 4.5 for cutting-edge performance" + "Superseded by Sonnet 4.5, 4.6, and Claude Sonnet 5" ], "best_for": [ @@ -587,6 +595,7 @@ ], "not_recommended_for": [ + "Any use — the model is retired and API requests fail; migrate to Claude Sonnet 4.6", "Latency-sensitive real-time applications", "Cost-sensitive high-volume workloads", "Tasks requiring latest training data (post-March 2025)", @@ -597,8 +606,8 @@ "pricing": { "input": "$3.00 per 1M tokens", "output": "$15.00 per 1M tokens", - "notes": "Batch API offers 50% discount. Prompt caching up to 90% savings.", - "last_verified": "2025-11-17" + "notes": "Historical pricing. Model retired 2026-06-15 — no longer purchasable on Anthropic-operated platforms.", + "last_verified": "2026-07-09" }, "context_window": 200000, "max_output_tokens": 64000, @@ -623,9 +632,10 @@ "training_cutoff": "March 2025" }, - "related_entities": ["claude-sonnet-4-5", "claude-opus-4", "claude-haiku-4-5", "gpt-5"], + "related_entities": ["claude-sonnet-4-6", "claude-sonnet-4-5", "claude-opus-4-8", "claude-haiku-4-5"], "tags": [ + "retired", "coding", "hipaa-eligible", "privacy", diff --git a/data/models/command-a-plus.json b/data/models/command-a-plus.json index f7d2594..814ce30 100644 --- a/data/models/command-a-plus.json +++ b/data/models/command-a-plus.json @@ -4,7 +4,7 @@ "name": "Command A+", "provider": "Cohere", "version": "20260520", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Cohere's Apache 2.0 open-weight 218B sparse MoE (25B active) unifying Command A, A Reasoning, A Vision, and A Translate. Runs on 2xH100 or a single B200, supports 48 languages, and ships native citations with grounding spans for verifiable RAG.", "website": "https://cohere.com/blog/command-a-plus", @@ -24,8 +24,8 @@ "value": "Solid coding performance; competitive with open peers but not the headline focus" } ], - "methodology": "Vendor-reported coding benchmarks; independent replication still emerging three weeks post-release", - "last_verified": "2026-06-10" + "methodology": "Vendor-reported coding benchmarks; independent replication still limited roughly seven weeks post-release", + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 87, @@ -39,7 +39,7 @@ } ], "methodology": "Vendor-reported reasoning benchmarks pending broad independent verification", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 88, @@ -53,7 +53,7 @@ } ], "methodology": "Review of vendor benchmarks emphasizing enterprise RAG, translation, and multilingual tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 88, @@ -67,7 +67,7 @@ } ], "methodology": "Review of grounded-generation behavior and enterprise consistency claims", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "2.0s", @@ -81,7 +81,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "4.5s", @@ -95,7 +95,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "256,000 tokens", @@ -109,7 +109,7 @@ } ], "methodology": "Provider documentation; A+ figure carried from Command A line pending explicit spec sheet", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 97, @@ -123,7 +123,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Released 2026-05-20, unifying Command A + A Reasoning + A Vision + A Translate into one model. Efficient serving (25B active, 2xH100 or 1xB200). Benchmarks are vendor-reported and independent verification is still emerging." @@ -144,7 +144,7 @@ } ], "methodology": "Review of vendor security documentation and enterprise deployment guidance against OWASP LLM01 patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 84, @@ -158,7 +158,7 @@ } ], "methodology": "Testing against adversarial prompt datasets; deployer-dependent for self-hosted use", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 88, @@ -172,7 +172,7 @@ } ], "methodology": "Analysis of deployment isolation options and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 86, @@ -186,7 +186,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 90, @@ -200,7 +200,7 @@ } ], "methodology": "Review of API security features, certifications, and enterprise controls", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strongest security posture among current open-weight releases thanks to Cohere's enterprise platform: SOC 2, VPC/private deployment, and a grounded-generation design that narrows injection surface in RAG." @@ -221,7 +221,7 @@ } ], "methodology": "Review of deployment documentation and residency options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -235,7 +235,7 @@ } ], "methodology": "Analysis of privacy policy and enterprise data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Configurable; zero retention available in private/VPC and self-hosted deployments", @@ -249,7 +249,7 @@ } ], "methodology": "Review of terms of service and deployment-dependent retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 86, @@ -263,7 +263,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 90, @@ -277,7 +277,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 90, @@ -291,7 +291,7 @@ } ], "methodology": "Review of self-hosting and private deployment options enabling zero retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Best-in-class privacy posture for an open-weight model: Western jurisdiction, SOC 2, and a full spectrum from SaaS to on-premises. The rare combination of open weights plus enterprise compliance is its core differentiator." @@ -312,7 +312,7 @@ } ], "methodology": "Evaluation of citation fidelity and grounding-span behavior in retrieval workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -326,7 +326,7 @@ } ], "methodology": "Testing on grounded QA workloads; ungrounded closed-book use performs closer to peers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 80, @@ -340,7 +340,7 @@ } ], "methodology": "Review of published bias evaluations and responsibility framework", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 84, @@ -354,7 +354,7 @@ } ], "methodology": "Assessment of confidence expression and grounding-span signaling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 88, @@ -368,7 +368,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -382,7 +382,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 84, @@ -396,7 +396,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms and configurability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Native citations and grounding spans are a genuine trust differentiator: outputs in RAG mode are verifiable against source passages by construction, not post-hoc." @@ -417,7 +417,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 88, @@ -431,7 +431,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 84, @@ -445,7 +445,7 @@ } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 84, @@ -459,7 +459,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -473,7 +473,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 86, @@ -487,7 +487,7 @@ } ], "methodology": "Analysis of third-party integrations, cloud availability, and tooling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 96, @@ -501,7 +501,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Mature enterprise platform behind a permissively licensed model. Apache 2.0 weights plus SOC 2 SaaS/VPC options give an unusually wide deployment spectrum." @@ -571,8 +571,8 @@ ], "limitations": [ - "Command A+ pricing not clearly published as of 2026-06-10 (prior Command A was $2.50/$10 per 1M)", - "Benchmarks vendor-reported; independent verification still emerging three weeks post-release", + "Command A+ API pricing still not published as of 2026-07-09 — per-token rates are 'contact sales' (prior Command A was $2.50/$10 per 1M)", + "Benchmarks vendor-reported; independent verification still limited roughly seven weeks post-release", "Behind agentic-coding specialists on SWE-bench-style tasks", "Training data sources not disclosed in detail", "Smaller open-source community than Chinese open-weight ecosystems" @@ -593,10 +593,10 @@ "metadata": { "pricing": { - "input": "Not yet published (prior Command A: $2.50 per 1M tokens)", - "output": "Not yet published (prior Command A: $10.00 per 1M tokens)", - "notes": "Command A+ API pricing not clearly published as of 2026-06-10; figures from the prior Command A generation shown for reference. Low confidence — verify with Cohere before procurement.", - "last_verified": "2026-06-10" + "input": "Not published (prior Command A: $2.50 per 1M tokens)", + "output": "Not published (prior Command A: $10.00 per 1M tokens)", + "notes": "Re-checked 2026-07-09: Cohere still publishes no per-token rate for Command A+ — production pricing is 'contact sales'. Figures from the prior Command A generation shown for reference only; verify with Cohere before procurement.", + "last_verified": "2026-07-09" }, "context_window": 256000, "languages": [ diff --git a/data/models/deepseek-r1.json b/data/models/deepseek-r1.json index c038094..1a4111c 100644 --- a/data/models/deepseek-r1.json +++ b/data/models/deepseek-r1.json @@ -4,9 +4,9 @@ "name": "DeepSeek-R1", "provider": "DeepSeek", "version": "20251020", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DeepSeek's standalone reasoning model, now superseded and discontinued as a product line. Its reasoning capabilities were folded into DeepSeek V3.1's hybrid thinking mode (Aug 2025), then V3.2 (Dec 2025) and V4 (Apr 2026); a successor 'R2' never shipped. The legacy deepseek-reasoner API endpoint is scheduled for deprecation on 2026-07-24. Historically achieved 53.6% on SWE-bench and 79.8% on HumanEval with strong coding and reasoning at competitive pricing.", + "description": "DeepSeek's standalone reasoning model, now superseded and discontinued. Its reasoning was folded into DeepSeek V3.1's hybrid thinking mode (Aug 2025), then V3.2 (Dec 2025) and V4 (Apr 2026); a successor 'R2' never shipped. The legacy deepseek-reasoner name no longer serves R1 — it routes to V4-Flash's thinking mode and is removed 2026-07-24 (15:59 UTC). R1 remains available only via third-party hosts or self-hosted open weights. Historically 53.6% SWE-bench and 79.8% HumanEval.", "website": "https://www.deepseek.com/", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring real-world software engineering tasks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 90, @@ -50,7 +50,7 @@ } ], "methodology": "Graduate-level reasoning benchmarks requiring multi-step problem solving", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 89, @@ -70,7 +70,7 @@ } ], "methodology": "Comprehensive knowledge testing across domains", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 88, @@ -84,7 +84,7 @@ } ], "methodology": "Internal testing with repeated prompts at various temperature settings", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.9s", @@ -98,7 +98,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.4s", @@ -112,7 +112,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "64,000 tokens", @@ -126,7 +126,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 97, @@ -140,7 +140,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong performance with excellent coding capabilities and efficient reasoning. Competitive latency despite reasoning optimization." @@ -160,7 +160,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 84, @@ -174,7 +174,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 80, @@ -188,7 +188,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 86, @@ -202,7 +202,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 84, @@ -216,7 +216,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good security posture with standard guardrails. Adequate protection for typical use cases." @@ -236,7 +236,7 @@ } ], "methodology": "Review of documentation and privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 88, @@ -250,7 +250,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "60 days (configurable)", @@ -264,7 +264,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 78, @@ -278,7 +278,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 82, @@ -292,7 +292,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -306,7 +306,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Moderate privacy posture. Data residency primarily in Asia. Limited compliance certifications for Western markets." @@ -326,7 +326,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -340,7 +340,7 @@ } ], "methodology": "Testing on factual QA datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 78, @@ -354,7 +354,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 79, @@ -368,7 +368,7 @@ } ], "methodology": "Assessment of confidence expression", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 85, @@ -382,7 +382,7 @@ } ], "methodology": "Review of documentation completeness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -396,7 +396,7 @@ } ], "methodology": "Review of public disclosures", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 85, @@ -410,7 +410,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Moderate transparency with standard safety features. Limited disclosure compared to Western providers." @@ -430,7 +430,7 @@ } ], "methodology": "Review of API design and consistency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 86, @@ -444,7 +444,7 @@ } ], "methodology": "Review of SDK quality and maintenance", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 74, @@ -461,11 +461,17 @@ "url": "https://api-docs.deepseek.com/quick_start/pricing", "date": "2026-06-10", "value": "Legacy deepseek-reasoner endpoint deprecates 2026-07-24; R1 line discontinued in favor of V3.1/V3.2/V4 hybrid thinking models" + }, + { + "source": "DeepSeek API Change Log", + "url": "https://api-docs.deepseek.com/updates/", + "date": "2026-07-09", + "value": "deepseek-reasoner alias currently routes to deepseek-v4-flash thinking mode (no longer serves R1); alias removed 2026-07-24 15:59 UTC, after which requests using it fail" } ], "methodology": "Review of versioning practices", - "last_verified": "2026-06-10", - "notes": "Model superseded; endpoint deprecation requires migration to V3.2/V4" + "last_verified": "2026-07-09", + "notes": "Model superseded; first-party endpoint no longer serves R1 weights. Migration to deepseek-v4-pro/-flash required before 2026-07-24" }, "monitoring_observability": { "score": 84, @@ -479,7 +485,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 85, @@ -493,7 +499,7 @@ } ], "methodology": "Assessment of support options", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 82, @@ -507,7 +513,7 @@ } ], "methodology": "Analysis of ecosystem maturity", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 93, @@ -521,7 +527,7 @@ } ], "methodology": "Review of licensing terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good operational quality with open licensing. Growing ecosystem with room for maturity." @@ -532,7 +538,7 @@ "overall": 92, "notes": "Excellent coding with 53.6% SWE-bench and 79.8% HumanEval. Strong value proposition with competitive pricing.", "alternatives": [ - "claude-4-sonnet", + "claude-sonnet-4", "openai-o1" ] }, @@ -540,23 +546,21 @@ "overall": 81, "notes": "Adequate for customer support but not specialized. Good latency helps.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "content-creation": { "overall": 83, "notes": "Solid content generation capabilities at competitive pricing.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "data-analysis": { "overall": 89, "notes": "Strong analytical capabilities with good reasoning. Excellent value for price.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -564,7 +568,7 @@ "overall": 87, "notes": "Good research capabilities with reasoning optimization.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -572,21 +576,21 @@ "overall": 78, "notes": "Limited compliance certifications for Western markets. Data residency concerns.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "healthcare": { "overall": 76, "notes": "Not suitable for healthcare due to limited compliance certifications and data residency.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "financial-analysis": { "overall": 86, "notes": "Good analytical capabilities at competitive pricing.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -594,7 +598,7 @@ "overall": 88, "notes": "Strong tutoring capabilities with good reasoning and affordable pricing.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -602,8 +606,7 @@ "overall": 82, "notes": "Adequate creative capabilities at good value.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] } }, @@ -622,7 +625,7 @@ "Limited transparency on training data", "Smaller context window (64K tokens)", "Less mature ecosystem compared to Western providers", - "Superseded: line discontinued in favor of DeepSeek V3.1/V3.2/V4 hybrid thinking; legacy deepseek-reasoner endpoint deprecates 2026-07-24" + "Superseded: line discontinued in favor of DeepSeek V3.1/V3.2/V4 hybrid thinking; legacy deepseek-reasoner alias now routes to V4-Flash (not R1) and is removed 2026-07-24, leaving third-party hosts or self-hosting as the only ways to run R1" ], "best_for": [ "Cost-sensitive projects needing strong coding capabilities", @@ -641,10 +644,10 @@ ], "metadata": { "pricing": { - "input": "$0.55 per 1M tokens", - "output": "$2.19 per 1M tokens", - "notes": "Highly competitive pricing, significantly lower than Western alternatives", - "last_verified": "2025-11-09" + "input": "$0.55 per 1M tokens (historical first-party pricing)", + "output": "$2.19 per 1M tokens (historical first-party pricing)", + "notes": "Historical first-party API pricing no longer applies: the deepseek-reasoner alias now serves V4-Flash at V4 pricing and is removed 2026-07-24. R1 is now served only by third-party hosts (rates vary) or self-hosted from open weights.", + "last_verified": "2026-07-09" }, "context_window": 64000, "languages": [ diff --git a/data/models/deepseek-v3-0324.json b/data/models/deepseek-v3-0324.json index e33a5fb..57cc6db 100644 --- a/data/models/deepseek-v3-0324.json +++ b/data/models/deepseek-v3-0324.json @@ -4,9 +4,9 @@ "name": "DeepSeek V3 0324", "provider": "DeepSeek", "version": "0324", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DeepSeek's March 2025 open-weights V3 checkpoint, now superseded by DeepSeek V3.1 (Aug 2025), V3.2 (Dec 2025), and V4 (Apr 2026). The legacy deepseek-chat endpoint name is scheduled for deprecation on 2026-07-24. Historically offered strong performance at competitive pricing for developers seeking capable models with transparent weights and commercial-friendly licensing.", + "description": "DeepSeek's March 2025 open-weights V3 checkpoint, now superseded by DeepSeek V3.1 (Aug 2025), V3.2 (Dec 2025), and V4 (Apr 2026). The legacy deepseek-chat name no longer serves this checkpoint — it routes to V4-Flash's non-thinking mode and is removed 2026-07-24 (15:59 UTC). V3-0324 remains available only via third-party hosts or self-hosted open weights. Historically offered strong performance at competitive pricing with transparent weights and commercial-friendly licensing.", "website": "https://www.deepseek.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 79, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 80, @@ -52,7 +52,7 @@ } ], "methodology": "Knowledge testing benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 77, @@ -66,7 +66,7 @@ } ], "methodology": "Community evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.2s", @@ -80,7 +80,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.6s", @@ -94,7 +94,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "64,000 tokens", @@ -108,7 +108,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 92, @@ -122,7 +122,7 @@ } ], "methodology": "Historical uptime", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong performance for open-weights model. Good balance of capabilities and cost." @@ -142,7 +142,7 @@ } ], "methodology": "Community adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 75, @@ -156,7 +156,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 78, @@ -170,7 +170,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 76, @@ -184,7 +184,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 78, @@ -198,7 +198,7 @@ } ], "methodology": "API security review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Moderate security for open-weights model. Additional safeguards recommended." @@ -218,7 +218,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 82, @@ -232,7 +232,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "90 days", @@ -246,7 +246,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 76, @@ -260,7 +260,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 72, @@ -274,7 +274,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 68, @@ -288,7 +288,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Standard privacy practices. 90-day retention longer than some competitors." @@ -308,7 +308,7 @@ } ], "methodology": "Reasoning evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 79, @@ -322,7 +322,7 @@ } ], "methodology": "Community evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 78, @@ -336,7 +336,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 80, @@ -350,7 +350,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 88, @@ -364,7 +364,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 85, @@ -378,7 +378,7 @@ } ], "methodology": "Technical documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 82, @@ -392,7 +392,7 @@ } ], "methodology": "Safety system analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency for open-weights model. Comprehensive documentation." @@ -412,7 +412,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 80, @@ -426,7 +426,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 72, @@ -443,11 +443,17 @@ "url": "https://api-docs.deepseek.com/quick_start/pricing", "date": "2026-06-10", "value": "Legacy deepseek-chat endpoint name deprecates 2026-07-24; superseded by V3.1, V3.2, and V4" + }, + { + "source": "DeepSeek API Change Log", + "url": "https://api-docs.deepseek.com/updates/", + "date": "2026-07-09", + "value": "deepseek-chat alias currently routes to deepseek-v4-flash non-thinking mode (no longer serves V3-0324); alias removed 2026-07-24 15:59 UTC, after which requests using it fail" } ], "methodology": "Versioning review", - "last_verified": "2026-06-10", - "notes": "Model superseded; endpoint deprecation requires migration to V3.2/V4" + "last_verified": "2026-07-09", + "notes": "Model superseded; first-party endpoint no longer serves this checkpoint. Migration to deepseek-v4-flash/-pro required before 2026-07-24" }, "monitoring_observability": { "score": 76, @@ -461,7 +467,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 78, @@ -475,7 +481,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 79, @@ -489,7 +495,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 88, @@ -503,7 +509,7 @@ } ], "methodology": "License review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good operational maturity with OpenAI-compatible API. Growing ecosystem." @@ -514,8 +520,7 @@ "overall": 81, "notes": "Good coding (59.4% HumanEval) at competitive pricing.", "alternatives": [ - "gpt-4-1", - "deepseek-coder" + "gpt-4-1" ] }, "customer-support": { @@ -599,7 +604,7 @@ "Smaller ecosystem than established providers", "Less mature enterprise support", "Smaller context window (64K tokens)", - "Superseded by DeepSeek V3.1/V3.2/V4; legacy deepseek-chat endpoint name deprecates 2026-07-24" + "Superseded by DeepSeek V3.1/V3.2/V4; legacy deepseek-chat alias now routes to V4-Flash (not this checkpoint) and is removed 2026-07-24, leaving third-party hosts or self-hosting as the only ways to run V3-0324" ], "best_for": [ "Cost-sensitive applications requiring good performance", @@ -618,10 +623,10 @@ ], "metadata": { "pricing": { - "input": "$0.27 per 1M tokens", - "output": "$1.09 per 1M tokens", - "notes": "Highly competitive pricing for performance level (cache miss pricing)", - "last_verified": "2025-11-09" + "input": "$0.27 per 1M tokens (historical first-party pricing)", + "output": "$1.09 per 1M tokens (historical first-party pricing)", + "notes": "Historical first-party API pricing no longer applies: the deepseek-chat alias now serves V4-Flash at V4 pricing and is removed 2026-07-24. V3-0324 is now served only by third-party hosts (rates vary) or self-hosted from open weights.", + "last_verified": "2026-07-09" }, "context_window": 64000, "languages": [ diff --git a/data/models/deepseek-v3-2.json b/data/models/deepseek-v3-2.json index 4f67a3f..843a9b9 100644 --- a/data/models/deepseek-v3-2.json +++ b/data/models/deepseek-v3-2.json @@ -4,7 +4,7 @@ "name": "DeepSeek-V3.2", "provider": "DeepSeek", "version": "20251201", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "DeepSeek's ~685B-parameter MoE flagship with DeepSeek Sparse Attention (DSA) for dramatically cheaper long-context inference. The V3.2-Speciale variant reached IMO 2025 gold-medal level (35/42) and 96.0% AIME. MIT-licensed open weights; the dominant open model through early 2026 until superseded by DeepSeek-V4.", "website": "https://www.deepseek.com/", @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks and competitive programming results from the official release and technical report", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -50,7 +50,7 @@ } ], "methodology": "Olympiad-level mathematics and competition benchmarks (IMO, AIME, ICPC) reported at release and corroborated by the technical report", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 91, @@ -64,7 +64,7 @@ } ], "methodology": "Comprehensive knowledge and instruction-following benchmark review from the technical report and community leaderboards", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 88, @@ -78,7 +78,7 @@ } ], "methodology": "Repeated-prompt testing across temperature settings and context lengths, supplemented by community reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.8s", @@ -92,7 +92,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes from independent benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "4.2s", @@ -106,7 +106,7 @@ } ], "methodology": "95th percentile response time across diverse workloads from independent benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -120,7 +120,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 96, @@ -134,7 +134,7 @@ } ], "methodology": "Historical uptime data from the official status page plus availability of multiple third-party hosts", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Frontier-level reasoning for an open model: V3.2-Speciale hit IMO 2025 gold-medal level and 2nd at ICPC World Finals. DSA makes 128K-context workloads unusually cheap. Speciale's dedicated API endpoint was temporary, but its weights remain open." @@ -154,7 +154,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attack patterns and community red-team reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 82, @@ -168,7 +168,7 @@ } ], "methodology": "Testing against adversarial prompt datasets; assessment accounts for open-weight modifiability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 80, @@ -182,7 +182,7 @@ } ], "methodology": "Analysis of privacy policy for the hosted API plus the self-hosting option for full data isolation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 85, @@ -196,7 +196,7 @@ } ], "methodology": "Safety testing across harmful content categories on default weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -210,7 +210,7 @@ } ], "methodology": "Review of API security features and transport guarantees", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Adequate default guardrails. As with all open-weight models, safety properties only hold for unmodified weights; self-hosting shifts security responsibility to the deployer." @@ -230,7 +230,7 @@ } ], "methodology": "Review of privacy policy and hosting options; China-jurisdiction caveat applies only to the first-party API", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 85, @@ -244,7 +244,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms for the hosted API", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Per first-party API policy (China-hosted); zero when self-hosted", @@ -258,7 +258,7 @@ } ], "methodology": "Review of terms of service; retention is deployment-dependent for open-weight models", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 76, @@ -272,7 +272,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 70, @@ -286,7 +286,7 @@ } ], "methodology": "Verification of certifications for the first-party platform; third-party hosted options inherit their providers' certifications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 80, @@ -300,7 +300,7 @@ } ], "methodology": "Review of data handling across first-party API, third-party hosts, and self-hosting", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard open-model split: the first-party API is China-hosted with China-jurisdiction data residency, while self-hosting or Western third-party hosts (which most regulated enterprises use) avoid that concern entirely." @@ -320,7 +320,7 @@ } ], "methodology": "Evaluation of reasoning transparency and trace accessibility", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 82, @@ -334,7 +334,7 @@ } ], "methodology": "Testing on factual QA datasets and community evaluations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 76, @@ -348,7 +348,7 @@ } ], "methodology": "Evaluation on bias benchmarks and politically sensitive topic probes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 78, @@ -362,7 +362,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -376,7 +376,7 @@ } ], "methodology": "Review of technical report and model card completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 75, @@ -390,7 +390,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 80, @@ -404,7 +404,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms in default weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong architectural transparency (open weights, detailed DSA technical report, visible reasoning traces) offset by limited training-data disclosure and known topic-avoidance on politically sensitive subjects." @@ -424,7 +424,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 84, @@ -438,7 +438,7 @@ } ], "methodology": "Review of SDK compatibility and inference-framework support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 78, @@ -452,7 +452,7 @@ } ], "methodology": "Review of versioning practices and historical endpoint lifecycle", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 82, @@ -466,7 +466,7 @@ } ], "methodology": "Review of monitoring tools across deployment options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 75, @@ -480,7 +480,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 90, @@ -494,7 +494,7 @@ } ], "methodology": "Analysis of third-party hosting, integrations, and community adoption", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 97, @@ -508,7 +508,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Mature ecosystem with broad third-party hosting. Main operational caveat is DeepSeek's rapid release cadence: V3.2 supersedes V3.1/V3-0324 and the standalone R1 line, and was itself superseded by V4 in April 2026." @@ -600,8 +600,8 @@ "pricing": { "input": "$0.28 per 1M tokens (first-party API, cache miss)", "output": "$0.42 per 1M tokens (first-party API)", - "notes": "DSA-driven price cut at launch made long-context usage exceptionally cheap; context caching discounts cache hits further. Legacy endpoints deprecate 2026-07-24 in favor of V4. Self-hosting cost is infrastructure-only under MIT license.", - "last_verified": "2026-06-10" + "notes": "DSA-driven price cut at launch made long-context usage exceptionally cheap; context caching discounts cache hits further. Confirmed July 2026: legacy first-party endpoints are removed 2026-07-24 (15:59 UTC) in favor of V4, after which V3.2 is served via third-party hosts or self-hosting only. Self-hosting cost is infrastructure-only under MIT license.", + "last_verified": "2026-07-09" }, "context_window": 128000, "max_output": 64000, diff --git a/data/models/deepseek-v4.json b/data/models/deepseek-v4.json index 0eee45b..37e43cb 100644 --- a/data/models/deepseek-v4.json +++ b/data/models/deepseek-v4.json @@ -4,9 +4,9 @@ "name": "DeepSeek-V4", "provider": "DeepSeek", "version": "20260424-preview", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DeepSeek's preview flagship family: V4-Pro (1.6T total / 49B active MoE, the largest open-weight release ever) and V4-Flash (284B/13B). 1M-token context with up to 384K output, built on manifold-constrained Hyper Connections and Constrained Sparse Attention. MIT license. Vendor benchmark claims await broad independent verification.", + "description": "DeepSeek's preview flagship family: V4-Pro (1.6T total / 49B active MoE, largest open-weight release ever) and V4-Flash (284B/13B). 1M context, up to 384K output, via manifold-constrained Hyper Connections and Constrained Sparse Attention. MIT license. Vendor benchmarks await broad independent verification. Per DeepSeek (2026-06-30), V4 graduates to official release mid-July 2026 — same model names, with peak-hour API pricing (2x baseline, Beijing 9:00-12:00 and 14:00-18:00).", "website": "https://www.deepseek.com/", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Vendor-reported coding benchmarks; medium confidence pending broad independent verification of preview-release claims", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 95, @@ -44,7 +44,7 @@ } ], "methodology": "Vendor-reported reasoning benchmarks; medium confidence until third-party evaluations of the preview mature", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 93, @@ -58,7 +58,7 @@ } ], "methodology": "Community leaderboard positions and vendor benchmarks for a preview release", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 85, @@ -72,7 +72,7 @@ } ], "methodology": "Repeated-prompt testing and community preview feedback", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "2.2s (V4-Pro), ~0.9s (V4-Flash)", @@ -86,7 +86,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes from independent benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "5.5s (V4-Pro)", @@ -100,7 +100,7 @@ } ], "methodology": "95th percentile response time across diverse workloads from independent benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -114,7 +114,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 92, @@ -128,7 +128,7 @@ } ], "methodology": "Status-page history since the 2026-04-24 launch", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Largest open-weight release ever (V4-Pro: 1.6T total / 49B active). Vendor benchmarks are impressive but this is a preview: score confidence is medium until independent verification matures. V4-Flash (284B/13B) offers a much cheaper deployment point." @@ -148,7 +148,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection patterns; limited preview-stage coverage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 82, @@ -162,7 +162,7 @@ } ], "methodology": "Adversarial prompt testing; assessment accounts for open-weight modifiability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 80, @@ -176,7 +176,7 @@ } ], "methodology": "Analysis of privacy policy plus self-hosting option for full data isolation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 84, @@ -190,7 +190,7 @@ } ], "methodology": "Safety testing across harmful content categories on default weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -204,7 +204,7 @@ } ], "methodology": "Review of API security features and transport guarantees", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Security posture mirrors V3.2 but with thinner preview-stage red-team coverage. Open weights shift safety responsibility to deployers who fine-tune or self-host." @@ -224,7 +224,7 @@ } ], "methodology": "Review of privacy policy and hosting options; China-jurisdiction caveat applies only to the first-party API", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 85, @@ -238,7 +238,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms for the hosted API", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Per first-party API policy (China-hosted); zero when self-hosted", @@ -252,7 +252,7 @@ } ], "methodology": "Review of terms of service; retention is deployment-dependent for open-weight models", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 76, @@ -266,7 +266,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 70, @@ -280,7 +280,7 @@ } ], "methodology": "Verification of certifications for the first-party platform; third-party hosts inherit their own certifications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 80, @@ -294,7 +294,7 @@ } ], "methodology": "Review of data handling across first-party API, third-party hosts, and self-hosting", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Same split as all DeepSeek releases: the first-party API is China-hosted with China-jurisdiction residency and no Western certifications, while self-hosting or Western third-party hosting avoids those concerns. V4-Pro's 1.6T size makes self-hosting far harder than V4-Flash." @@ -314,7 +314,7 @@ } ], "methodology": "Evaluation of reasoning transparency and trace accessibility", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -328,7 +328,7 @@ } ], "methodology": "Limited factual QA testing during the preview period", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 75, @@ -342,7 +342,7 @@ } ], "methodology": "Preliminary bias probing; formal evaluations pending", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 78, @@ -356,7 +356,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 85, @@ -370,7 +370,7 @@ } ], "methodology": "Review of preview documentation completeness against DeepSeek's historical technical-report standard", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -384,7 +384,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 79, @@ -398,7 +398,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms in default weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Architectural novelty (manifold-constrained Hyper Connections, Constrained Sparse Attention) is disclosed, but the preview lacks the full technical report and independent benchmark replication DeepSeek usually delivers. Treat vendor claims with medium confidence." @@ -418,7 +418,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 84, @@ -432,7 +432,7 @@ } ], "methodology": "Review of SDK compatibility and inference-framework support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 74, @@ -443,10 +443,16 @@ "url": "https://api-docs.deepseek.com/quick_start/pricing", "date": "2026-04-24", "value": "Legacy deepseek-chat/deepseek-reasoner endpoints deprecate 2026-07-24, a three-month migration window" + }, + { + "source": "TechNode", + "url": "https://technode.com/2026/06/30/deepseek-to-launch-v4-in-mid-july-with-new-peak-time-api-pricing/", + "date": "2026-06-30", + "value": "Official V4 release planned for mid-July 2026 (graduation from preview, same model names) with peak-hour pricing at 2x baseline during Beijing 9:00-12:00 and 14:00-18:00; users get 24-hour email notice before billing changes" } ], "methodology": "Review of deprecation timelines and migration windows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 82, @@ -460,7 +466,7 @@ } ], "methodology": "Review of monitoring tools across deployment options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 74, @@ -474,7 +480,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 84, @@ -488,7 +494,7 @@ } ], "methodology": "Analysis of third-party hosting availability six weeks post-release", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 97, @@ -502,7 +508,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Aggressive pricing (V4-Pro $0.435/$0.87, V4-Flash $0.14/$0.28 per 1M) and MIT licensing, but the short legacy-endpoint deprecation window (2026-07-24) and preview status demand migration agility. V4-Pro self-hosting is feasible only for well-resourced organizations." @@ -569,13 +575,13 @@ "Inherits DeepSeek's frontier reasoning lineage from V3.2/Speciale" ], "limitations": [ - "Preview status: behavior, pricing, and endpoints may change; vendor benchmark claims not yet broadly independently verified", + "Preview status until mid-July 2026 official release; vendor benchmark claims not yet broadly independently verified and full technical report not yet published", + "Peak-hour API pricing (2x baseline, Beijing 9:00-12:00 and 14:00-18:00) takes effect at the official release, complicating cost planning for workloads in those windows", "First-party API is China-hosted: China-jurisdiction data residency and no SOC 2/HIPAA/FedRAMP (self-hosting or Western hosts avoid this)", "V4-Pro's 1.6T footprint makes self-hosting impractical for all but the largest organizations", "Short migration window: legacy deepseek-chat/deepseek-reasoner endpoints deprecate 2026-07-24", "Text-only: no native vision or audio", - "No enterprise SLA or dedicated support on the first-party platform", - "Full technical report not yet published for the preview" + "No enterprise SLA or dedicated support on the first-party platform" ], "best_for": [ "Million-token context workloads: whole-repository coding, corpus-scale research, multi-document analysis", @@ -593,8 +599,8 @@ "pricing": { "input": "$0.435 per 1M tokens (V4-Pro); $0.14 per 1M (V4-Flash, $0.0028 cache hit)", "output": "$0.87 per 1M tokens (V4-Pro); $0.28 per 1M (V4-Flash)", - "notes": "Preview pricing per official pricing page; legacy deepseek-chat/deepseek-reasoner endpoints deprecate 2026-07-24. Self-hosting is infrastructure-cost-only under MIT license.", - "last_verified": "2026-06-10" + "notes": "Preview pricing per official pricing page; legacy deepseek-chat/deepseek-reasoner endpoints deprecate 2026-07-24. At the mid-July 2026 official release, peak-hour pricing takes effect: 2x baseline during Beijing 9:00-12:00 and 14:00-18:00 (off-peak rates unchanged). Self-hosting is infrastructure-cost-only under MIT license.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 384000, diff --git a/data/models/gemini-2-0-flash.json b/data/models/gemini-2-0-flash.json index 6c01aab..d4f358c 100644 --- a/data/models/gemini-2-0-flash.json +++ b/data/models/gemini-2-0-flash.json @@ -4,7 +4,7 @@ "name": "Gemini 2.0 Flash", "provider": "Google", "version": "20251101", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "SHUT DOWN: Google shut down the Gemini 2.0 Flash family on 2026-06-01; the model is no longer served via the Gemini API. Historically a fast, efficient multimodal model (53.6% SWE-bench, 62.1% MMLU) optimized for speed and vision. Migrate to Gemini 3.5 Flash for equivalent fast multimodal workloads.", "website": "https://deepmind.google/technologies/gemini/", @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 86, @@ -50,7 +50,7 @@ } ], "methodology": "Graduate-level reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 85, @@ -64,7 +64,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 87, @@ -78,7 +78,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.8s", @@ -92,7 +92,7 @@ } ], "methodology": "Median latency for API requests", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "1.5s", @@ -106,7 +106,7 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -120,7 +120,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -134,7 +134,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Fast and efficient with excellent multimodal capabilities. 1M token context window is industry-leading. Optimized for speed with 0.8s p50 latency." @@ -154,7 +154,7 @@ } ], "methodology": "Testing against OWASP LLM01", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 89, @@ -168,7 +168,7 @@ } ], "methodology": "Testing against adversarial prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 84, @@ -182,7 +182,7 @@ } ], "methodology": "Analysis of privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 90, @@ -196,7 +196,7 @@ } ], "methodology": "Safety testing across content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 88, @@ -210,7 +210,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong security leveraging Google Cloud infrastructure. Comprehensive safety guardrails." @@ -230,7 +230,7 @@ } ], "methodology": "Review of Google Cloud documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -244,7 +244,7 @@ } ], "methodology": "Analysis of privacy policy", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral for API)", @@ -258,7 +258,7 @@ } ], "methodology": "Review of data retention policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 85, @@ -272,7 +272,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 92, @@ -286,7 +286,7 @@ } ], "methodology": "Verification of certifications", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 93, @@ -300,7 +300,7 @@ } ], "methodology": "Review of data handling", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent privacy leveraging Google Cloud infrastructure. HIPAA eligible with comprehensive certifications." @@ -320,7 +320,7 @@ } ], "methodology": "Evaluation of explanation features", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 83, @@ -334,7 +334,7 @@ } ], "methodology": "Testing on factual QA datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -348,7 +348,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 81, @@ -362,7 +362,7 @@ } ], "methodology": "Assessment of confidence expression", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 88, @@ -376,7 +376,7 @@ } ], "methodology": "Review of documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 77, @@ -390,7 +390,7 @@ } ], "methodology": "Review of public disclosures", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -404,7 +404,7 @@ } ], "methodology": "Analysis of safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with strong safety guardrails. Standard explainability for a Flash model." @@ -424,7 +424,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -438,7 +438,7 @@ } ], "methodology": "Review of SDK quality", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 79, @@ -455,10 +455,16 @@ "url": "https://ai.google.dev/gemini-api/docs/changelog", "date": "2026-06-10", "value": "Gemini 2.0 Flash family shut down 2026-06-01; no longer served" + }, + { + "source": "Gemini API Deprecations", + "url": "https://ai.google.dev/gemini-api/docs/deprecations", + "date": "2026-07-09", + "value": "Shutdown confirmed: gemini-2.0-flash, -001, -lite, -lite-001 all retired 2026-06-01; designated replacement is gemini-3.5-flash" } ], "methodology": "Review of versioning practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 92, @@ -472,7 +478,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -486,7 +492,7 @@ } ], "methodology": "Assessment of support options", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 83, @@ -500,7 +506,7 @@ } ], "methodology": "Analysis of ecosystem", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -514,7 +520,7 @@ } ], "methodology": "Review of licensing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Model shut down 2026-06-01 and no longer served. Versioning and ecosystem scores reduced to reflect shutdown." @@ -525,7 +531,7 @@ "overall": 90, "notes": "Strong coding with 53.6% SWE-bench. Fast latency ideal for interactive development.", "alternatives": [ - "claude-4-sonnet", + "claude-sonnet-4", "openai-o1" ] }, @@ -533,22 +539,21 @@ "overall": 92, "notes": "Excellent for customer support with fast 0.8s latency. Multimodal capabilities help with image support.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "content-creation": { "overall": 86, "notes": "Good content generation with fast turnaround. Strong for high-volume content needs.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "data-analysis": { "overall": 88, "notes": "Good analytical capabilities with fast processing. 1M context window excellent for large datasets.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -556,7 +561,7 @@ "overall": 87, "notes": "Good research capabilities with massive 1M context for large documents.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -564,21 +569,21 @@ "overall": 86, "notes": "HIPAA eligible with strong Google Cloud compliance. Good for regulated industries.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "healthcare": { "overall": 85, "notes": "HIPAA eligible with comprehensive compliance certifications.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "financial-analysis": { "overall": 87, "notes": "Good analytical capabilities with fast processing for real-time analysis.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -586,7 +591,7 @@ "overall": 89, "notes": "Strong for education with fast responses and multimodal capabilities.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -594,8 +599,7 @@ "overall": 84, "notes": "Adequate creative capabilities with fast generation.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] } }, @@ -633,8 +637,8 @@ "pricing": { "input": "$0.10 per 1M tokens", "output": "$0.40 per 1M tokens", - "notes": "Highly competitive Flash tier pricing - exceptional value for speed", - "last_verified": "2025-11-09" + "notes": "Historical pricing; model shut down 2026-06-01 and is no longer purchasable. Replacements cost more (gemini-3.5-flash is $1.50/$9.00).", + "last_verified": "2026-07-09" }, "context_window": 1000000, "languages": [ diff --git a/data/models/gemini-2-5-pro.json b/data/models/gemini-2-5-pro.json index 53e68d4..bd7823c 100644 --- a/data/models/gemini-2-5-pro.json +++ b/data/models/gemini-2-5-pro.json @@ -4,9 +4,9 @@ "name": "Gemini 2.5 Pro", "provider": "Google", "version": "gemini-2.5-pro-002", - "last_evaluated": "2025-11-07", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Google's latest flagship with 2M token context window, Deep Think mode for complex reasoning, and native multimodal capabilities. Best-in-class for long-context applications.", + "description": "DEPRECATED: Google has scheduled Gemini 2.5 Pro for shutdown on 2026-10-16; the designated replacement is Gemini 3.1 Pro. Formerly Google's flagship (superseded by the Gemini 3.x line since late 2025), with 1M token context window (2M on select Vertex AI enterprise tiers), Deep Think mode for complex reasoning, and native multimodal capabilities. Do not start new projects on this model.", "website": "https://deepmind.google/technologies/gemini/", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Standard coding benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 94, @@ -39,7 +39,7 @@ } ], "methodology": "PhD-level reasoning benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 93, @@ -53,7 +53,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 92, @@ -67,7 +67,7 @@ } ], "methodology": "Internal consistency testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.5s", @@ -81,7 +81,7 @@ } ], "methodology": "API latency measurements", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.5s", @@ -95,21 +95,21 @@ } ], "methodology": "95th percentile measurements", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "context_window": { - "value": "2,000,000 tokens", + "value": "1,000,000 tokens (2M on select Vertex AI enterprise tiers)", "confidence": "high", "evidence": [ { "source": "Google AI Documentation", - "url": "https://ai.google.dev/gemini-api/docs/models/gemini", - "date": "2025-02-20", - "value": "2M token context window (largest available)" + "url": "https://ai.google.dev/gemini-api/docs/long-context", + "date": "2026-07-09", + "value": "1M token context window on the standard Gemini API; 2M extension limited to select Vertex AI enterprise tiers and never rolled out to the standard API" } ], "methodology": "Official specification", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uptime": { "score": 97, @@ -123,10 +123,10 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, - "notes": "Exceptional performance with 2M context window enabling unprecedented long-document processing. Deep Think mode enhances complex reasoning." + "notes": "Strong performance for its generation with 1M context window and Deep Think mode, but superseded by the Gemini 3.x line. DEPRECATED: shutdown scheduled 2026-10-16." }, "security": { @@ -144,7 +144,7 @@ } ], "methodology": "OWASP LLM security testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 87, @@ -158,7 +158,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -172,7 +172,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_safety": { "score": 89, @@ -186,7 +186,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "api_security": { "score": 87, @@ -200,7 +200,7 @@ } ], "methodology": "API security review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Strong security leveraging Google Cloud infrastructure. Configurable safety filters provide flexibility." @@ -221,7 +221,7 @@ } ], "methodology": "Cloud infrastructure review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 88, @@ -235,7 +235,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Varies by tier", @@ -249,7 +249,7 @@ } ], "methodology": "Data retention policy review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Enterprise options available for zero retention" }, "pii_handling": { @@ -264,7 +264,7 @@ } ], "methodology": "Data protection review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 90, @@ -278,7 +278,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 83, @@ -292,7 +292,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Good privacy with Google Cloud infrastructure. Enterprise options provide enhanced controls. HIPAA compliance available through Google Cloud." @@ -313,7 +313,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 86, @@ -327,7 +327,7 @@ } ], "methodology": "Factual QA testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 83, @@ -341,7 +341,7 @@ } ], "methodology": "Bias benchmark evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 85, @@ -355,7 +355,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 90, @@ -369,7 +369,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 82, @@ -383,7 +383,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -397,7 +397,7 @@ } ], "methodology": "Safety mechanism review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency with Deep Think mode and comprehensive documentation. Configurable guardrails provide flexibility." @@ -418,7 +418,7 @@ } ], "methodology": "API design review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -432,10 +432,10 @@ } ], "methodology": "SDK quality assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 91, + "score": 85, "confidence": "high", "evidence": [ { @@ -443,10 +443,16 @@ "url": "https://cloud.google.com/apis/design/versioning", "date": "2025-01-01", "value": "Clear versioning with migration guides" + }, + { + "source": "Gemini API Deprecations", + "url": "https://ai.google.dev/gemini-api/docs/deprecations", + "date": "2026-07-09", + "value": "gemini-2.5-pro scheduled for shutdown 2026-10-16 (extended from an original June 2026 date); designated replacement is gemini-3.1-pro" } ], "methodology": "Versioning policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -460,7 +466,7 @@ } ], "methodology": "Observability tools review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "support_quality": { "score": 92, @@ -474,7 +480,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 90, @@ -488,7 +494,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -502,10 +508,10 @@ } ], "methodology": "License review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, - "notes": "Excellent operational maturity backed by Google Cloud infrastructure. Best-in-class monitoring and observability." + "notes": "Excellent operational maturity backed by Google Cloud infrastructure, but the model is deprecated with a 2026-10-16 shutdown date. Versioning score reduced to reflect the pending retirement; migrate to Gemini 3.1 Pro." } }, @@ -527,12 +533,12 @@ }, "data-analysis": { "overall": 95, - "notes": "Outstanding for data analysis with 2M context enabling analysis of massive datasets.", + "notes": "Outstanding for data analysis with 1M context enabling analysis of massive datasets.", "alternatives": ["claude-opus-4-1"] }, "research-assistant": { "overall": 96, - "notes": "Exceptional for research with 2M context. Can process entire books, papers, and repositories.", + "notes": "Exceptional for research with 1M context. Can process entire books, papers, and repositories.", "alternatives": ["claude-opus-4-1"] }, "legal-compliance": { @@ -563,7 +569,7 @@ }, "strengths": [ - "2M token context window - largest available (10x Claude, 16x GPT-5)", + "1M token context window (2M on select Vertex AI enterprise tiers)", "Deep Think mode for enhanced reasoning on complex problems", "Native multimodal capabilities (text, image, video, audio)", "Google Cloud infrastructure with enterprise-grade reliability", @@ -573,10 +579,11 @@ "limitations": [ "Slightly behind Claude/GPT on specialized benchmarks", - "Newer model with less community testing", "Deep Think mode increases latency significantly", "Data retention policies less transparent than Anthropic", - "Smaller ecosystem than OpenAI" + "Smaller ecosystem than OpenAI", + "Superseded by the Gemini 3.x line (Gemini 3.1 Pro is the current Pro tier)", + "DEPRECATED: shutdown scheduled 2026-10-16; migrate to Gemini 3.1 Pro" ], "best_for": [ @@ -588,6 +595,7 @@ ], "not_recommended_for": [ + "New projects of any kind (model shuts down 2026-10-16; start on Gemini 3.1 Pro instead)", "Applications requiring lowest possible latency", "Highly regulated industries prioritizing zero-retention (without enterprise setup)", "Projects requiring OpenAI-specific ecosystem features" @@ -597,10 +605,10 @@ "pricing": { "input": "$1.25 per 1M tokens (<200k context), $2.50 per 1M tokens (>200k context)", "output": "$10.00 per 1M tokens (<200k context), $15.00 per 1M tokens (>200k context)", - "notes": "Tiered pricing based on context length - most cost-effective frontier model for long-context work", - "last_verified": "2025-11-09" + "notes": "Tiered pricing based on context length. Confirmed unchanged on the official pricing page as of 2026-07-09; billing ends when the model shuts down 2026-10-16.", + "last_verified": "2026-07-09" }, - "context_window": 2000000, + "context_window": 1000000, "languages": [ "English", "100+ languages" @@ -612,11 +620,12 @@ "parameters": "Not disclosed" }, - "related_entities": ["claude-opus-4-1", "gpt-5", "claude-sonnet-4-5"], + "related_entities": ["gemini-3-1-pro", "gemini-3-5-flash", "claude-opus-4-1", "gpt-5", "claude-sonnet-4-5"], "tags": [ + "deprecated", "long-context", - "2m-tokens", + "1m-tokens", "deep-think", "multimodal", "cost-effective", diff --git a/data/models/gemini-3-1-pro.json b/data/models/gemini-3-1-pro.json index 2306cbf..586775e 100644 --- a/data/models/gemini-3-1-pro.json +++ b/data/models/gemini-3-1-pro.json @@ -3,10 +3,10 @@ "type": "model", "name": "Gemini 3.1 Pro", "provider": "Google", - "version": "gemini-3.1-pro", - "last_evaluated": "2026-06-10", + "version": "gemini-3.1-pro-preview", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Google's GA flagship reasoning model with 77.1% ARC-AGI-2 (2.5x Gemini 3 Pro), 94.3% GPQA Diamond, 2887 Elo on LiveCodeBench Pro, and 1M token context. Supersedes Gemini 3 Pro Preview as the production-ready frontier tier.", + "description": "Google's current flagship reasoning model with 77.1% ARC-AGI-2 (2.5x Gemini 3 Pro), 94.3% GPQA Diamond, 2887 Elo on LiveCodeBench Pro, and 1M token context. Supersedes the retired Gemini 3 Pro Preview (shut down 2026-03-09; the gemini-pro-latest alias now points here). Note: still served under the preview model ID gemini-3.1-pro-preview — official docs do not list it as GA, contrary to earlier reports. Gemini 3.5 Pro (announced I/O May 2026) has not shipped as of 2026-07-09.", "website": "https://ai.google.dev/gemini-api/docs/changelog", "trust_vector": { @@ -31,7 +31,7 @@ } ], "methodology": "Competitive programming and agentic tool-use benchmarks from official launch materials", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 98, @@ -50,8 +50,8 @@ "value": "94.3% on PhD-level science questions" } ], - "methodology": "Abstract reasoning and PhD-level science benchmarks reported at GA launch", - "last_verified": "2026-06-10" + "methodology": "Abstract reasoning and PhD-level science benchmarks reported at launch", + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 96, @@ -61,11 +61,11 @@ "source": "Google DeepMind Models Page", "url": "https://deepmind.google/models/gemini/", "date": "2026-02-19", - "value": "Positioned as GA flagship, superseding Gemini 3 Pro Preview across general benchmarks" + "value": "Positioned as flagship Pro tier, superseding Gemini 3 Pro Preview across general benchmarks" } ], "methodology": "Cross-benchmark comparison against predecessor Gemini 3 Pro Preview", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 94, @@ -75,11 +75,11 @@ "source": "Google AI Documentation", "url": "https://ai.google.dev/gemini-api/docs", "date": "2026-02-19", - "value": "GA stability commitments versus preview-tier predecessor" + "value": "Stable flagship serving since 2026-02-19, though the model ID remains gemini-3.1-pro-preview" } ], - "methodology": "Consistency assessment based on GA status and documented model behavior", - "last_verified": "2026-06-10" + "methodology": "Consistency assessment based on serving track record and documented model behavior", + "last_verified": "2026-07-09" }, "latency_p50": { "value": "~1.8s", @@ -93,7 +93,7 @@ } ], "methodology": "Median latency from third-party aggregator measurements", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -103,11 +103,11 @@ "source": "Gemini API Changelog", "url": "https://ai.google.dev/gemini-api/docs/changelog", "date": "2026-02-19", - "value": "1M token context window at GA" + "value": "1M token context window at launch" } ], "methodology": "Official specification from provider documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -121,10 +121,10 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Massive reasoning jump: 77.1% ARC-AGI-2 vs 31.1% for Gemini 3 Pro. GA status removes preview-tier risk. Note: Gemini 3.5 Pro was announced at I/O May 2026 but is not yet GA." + "notes": "Massive reasoning jump: 77.1% ARC-AGI-2 vs 31.1% for Gemini 3 Pro. Correction 2026-07-09: official docs list the model as gemini-3.1-pro-preview (preview, not GA), contrary to the earlier GA characterization; it is nonetheless Google's designated migration target for retired/deprecated Pro models. Gemini 3.5 Pro (announced I/O May 2026, June GA target slipped) has still not shipped as of 2026-07-09." }, "security": { @@ -142,7 +142,7 @@ } ], "methodology": "OWASP LLM01 prompt injection testing and vendor safety documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -156,7 +156,7 @@ } ], "methodology": "Adversarial prompt dataset testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -170,7 +170,7 @@ } ], "methodology": "Privacy policy and API terms review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 90, @@ -184,7 +184,7 @@ } ], "methodology": "Safety filter testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 89, @@ -198,7 +198,7 @@ } ], "methodology": "Review of API security features and infrastructure", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Inherits Google Cloud security posture. Configurable safety filters and Vertex AI IAM controls for enterprise deployment." @@ -219,7 +219,7 @@ } ], "methodology": "Cloud infrastructure and data residency documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -233,7 +233,7 @@ } ], "methodology": "Terms of service review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Varies by tier; zero retention available on Vertex AI enterprise", @@ -247,7 +247,7 @@ } ], "methodology": "Data retention policy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 84, @@ -261,7 +261,7 @@ } ], "methodology": "Data protection capability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 92, @@ -275,7 +275,7 @@ } ], "methodology": "Certification verification through Google Cloud compliance center", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 86, @@ -289,7 +289,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong enterprise posture via Vertex AI data governance, SOC/ISO certifications, and EU data residency options." @@ -310,7 +310,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -324,7 +324,7 @@ } ], "methodology": "Factual QA testing and vendor claims review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -338,7 +338,7 @@ } ], "methodology": "Bias benchmark evaluation and policy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 87, @@ -352,7 +352,7 @@ } ], "methodology": "Qualitative assessment of confidence expression", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -362,11 +362,11 @@ "source": "Gemini Documentation", "url": "https://ai.google.dev/gemini-api/docs/changelog", "date": "2026-02-19", - "value": "Comprehensive GA documentation with benchmarks and limitations" + "value": "Comprehensive launch documentation with benchmarks and limitations" } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 82, @@ -380,7 +380,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 91, @@ -394,10 +394,10 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Strong transparency via exposed thinking traces and comprehensive GA documentation. Training data details remain limited (industry standard)." + "notes": "Strong transparency via exposed thinking traces and comprehensive documentation. Training data details remain limited (industry standard)." }, "operational_excellence": { @@ -415,7 +415,7 @@ } ], "methodology": "API design and feature completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -429,7 +429,7 @@ } ], "methodology": "SDK quality and maintenance assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 92, @@ -439,11 +439,17 @@ "source": "Gemini API Changelog", "url": "https://ai.google.dev/gemini-api/docs/changelog", "date": "2026-02-19", - "value": "GA release 2026-02-19 with documented deprecation timeline for 3 Pro Preview" + "value": "Released 2026-02-19 with documented deprecation timeline for 3 Pro Preview" + }, + { + "source": "Gemini API Deprecations", + "url": "https://ai.google.dev/gemini-api/docs/deprecations", + "date": "2026-07-09", + "value": "gemini-3.1-pro-preview active with no shutdown date; designated replacement for gemini-3-pro-preview (shut down 2026-03-09) and gemini-2.5-pro (shutdown 2026-10-16); gemini-pro-latest alias points here since 2026-03-06" } ], "methodology": "Versioning policy and changelog review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 95, @@ -457,7 +463,7 @@ } ], "methodology": "Observability tooling review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 93, @@ -471,7 +477,7 @@ } ], "methodology": "Support channel assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 93, @@ -485,7 +491,7 @@ } ], "methodology": "Ecosystem and integration analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -499,24 +505,24 @@ } ], "methodology": "License terms review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pricing_transparency": { - "value": "~$2/$12 per 1M tokens (<=200K context), $4/$18 above 200K", - "confidence": "medium", + "value": "$2/$12 per 1M tokens (<=200K context), $4/$18 above 200K", + "confidence": "high", "evidence": [ { - "source": "Pricing aggregators", - "url": "https://artificialanalysis.ai/", - "date": "2026-06-01", - "value": "Aggregator-sourced: ~$2 input / $12 output per 1M tokens at <=200K context; $4/$18 above" + "source": "Gemini API Pricing", + "url": "https://ai.google.dev/gemini-api/docs/pricing", + "date": "2026-07-09", + "value": "Official: $2.00 input / $12.00 output per 1M tokens at <=200K context; $4.00/$18.00 above 200K; context caching $0.20-$0.40 plus $4.50/hour storage; paid tier only (no free tier)" } ], - "methodology": "Cross-referenced third-party pricing aggregators; official pricing page recommended for confirmation", - "last_verified": "2026-06-10" + "methodology": "Verified against the official Gemini API pricing page", + "last_verified": "2026-07-09" } }, - "notes": "Mature GA operational posture across all Google AI surfaces. Pricing figures are aggregator-sourced (medium confidence)." + "notes": "Mature operational posture across all Google AI surfaces since 2026-02-19 launch. Pricing confirmed on the official pricing page (2026-07-09). Model ID remains gemini-3.1-pro-preview despite flagship positioning." } }, @@ -567,16 +573,17 @@ "Exceptional abstract reasoning: 77.1% ARC-AGI-2 (~2.5x Gemini 3 Pro's 31.1%)", "94.3% GPQA Diamond, near-saturation PhD-level science", "Frontier coding: 2887 Elo LiveCodeBench Pro, 78.2% MCP Atlas", - "1M token context window with GA stability", + "1M token context window; Google's designated migration target for the retired 3 Pro Preview and deprecated 2.5 Pro", "Enterprise posture: Vertex AI data governance, SOC/ISO certs, EU residency", "Day-one availability across AI Studio, Vertex AI, and Gemini app" ], "limitations": [ - "Pricing confidence medium: figures aggregator-sourced (~$2/$12, $4/$18 above 200K)", + "Still served under a preview model ID (gemini-3.1-pro-preview); not listed as GA in official docs", + "Paid tier only — no free tier access (unique among current Gemini API models)", "Extended reasoning modes add significant latency", "Training data transparency limited (industry standard)", - "Gemini 3.5 Pro announced at I/O May 2026 may supersede it soon (not yet GA)", + "Gemini 3.5 Pro (announced I/O May 2026, GA target slipped past June) may supersede it soon", "Long-context (>200K) pricing roughly doubles per-token cost" ], @@ -590,15 +597,15 @@ "not_recommended_for": [ "Latency-sensitive applications (consider Gemini 3.5 Flash)", "High-volume cost-sensitive inference", - "Teams needing fully confirmed first-party pricing before procurement" + "Free-tier prototyping (model is paid-only)" ], "metadata": { "pricing": { - "input": "~$2.00 per 1M tokens (<=200K), ~$4.00 per 1M tokens (>200K)", - "output": "~$12.00 per 1M tokens (<=200K), ~$18.00 per 1M tokens (>200K)", - "notes": "Aggregator-sourced (medium confidence). Tiered by context length; verify against official pricing page.", - "last_verified": "2026-06-10" + "input": "$2.00 per 1M tokens (<=200K), $4.00 per 1M tokens (>200K)", + "output": "$12.00 per 1M tokens (<=200K), $18.00 per 1M tokens (>200K)", + "notes": "Confirmed on the official Gemini API pricing page 2026-07-09. Context caching $0.20-$0.40 per 1M plus $4.50/hour storage. Paid tier only (no free tier). Same pricing as the retired Gemini 3 Pro.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 64000, @@ -616,7 +623,7 @@ "tags": [ "flagship", - "ga", + "preview", "reasoning", "long-context", "1m-tokens", diff --git a/data/models/gemini-3-5-flash.json b/data/models/gemini-3-5-flash.json index 70b8d4b..baff799 100644 --- a/data/models/gemini-3-5-flash.json +++ b/data/models/gemini-3-5-flash.json @@ -4,7 +4,7 @@ "name": "Gemini 3.5 Flash", "provider": "Google", "version": "gemini-3.5-flash", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Google's GA 'frontier workhorse' launched at I/O 2026. Beats Gemini 3.1 Pro on agentic and coding suites (76.2% Terminal-Bench 2.1, 83.6% MCP Atlas) at roughly 4x the speed, with 1M token context. Pricier than past Flash tiers at $1.50/$9.00 per 1M.", "website": "https://deepmind.google/models/gemini/", @@ -31,7 +31,7 @@ } ], "methodology": "Official launch benchmarks for agentic coding; vendor-reported, pending broad third-party replication", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 89, @@ -51,7 +51,7 @@ } ], "methodology": "Official agentic and multimodal reasoning benchmarks from launch materials", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 91, @@ -65,7 +65,7 @@ } ], "methodology": "Cross-benchmark comparison against Gemini 3.1 Pro from official launch claims", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 92, @@ -79,7 +79,7 @@ } ], "methodology": "Consistency assessment based on GA status and vendor throughput claims", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "<1s", @@ -93,7 +93,7 @@ } ], "methodology": "Relative speed claims from official materials; absolute latency varies by workload", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,048,576 tokens input / 65,536 output", @@ -107,7 +107,7 @@ } ], "methodology": "Official specification from provider documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -121,7 +121,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Unusual positioning: a Flash-tier model that beats the flagship 3.1 Pro on agentic/coding suites at ~4x speed. Benchmark figures are official claims from launch (2026-05-19); third-party replication still maturing." @@ -142,7 +142,7 @@ } ], "methodology": "OWASP LLM01 testing and vendor documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 86, @@ -156,7 +156,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -170,7 +170,7 @@ } ], "methodology": "Privacy policy and API terms review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 89, @@ -184,7 +184,7 @@ } ], "methodology": "Safety filter testing across content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 89, @@ -198,7 +198,7 @@ } ], "methodology": "API security feature review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard Gemini 3.x-family security posture on Google Cloud infrastructure. High-speed agentic use increases the importance of downstream tool sandboxing." @@ -219,7 +219,7 @@ } ], "methodology": "Cloud infrastructure documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -233,7 +233,7 @@ } ], "methodology": "Terms of service review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Varies by tier; zero retention available on Vertex AI enterprise", @@ -247,7 +247,7 @@ } ], "methodology": "Data retention policy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 84, @@ -261,7 +261,7 @@ } ], "methodology": "Data protection capability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 92, @@ -275,7 +275,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 86, @@ -289,7 +289,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Same Google Cloud compliance envelope as the Pro tier: SOC/ISO certifications, GDPR, HIPAA via Google Cloud, EU residency options." @@ -310,7 +310,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 84, @@ -324,7 +324,7 @@ } ], "methodology": "Benchmark-derived grounding assessment; vendor claims pending independent replication", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -338,7 +338,7 @@ } ], "methodology": "Bias benchmark evaluation and policy review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 84, @@ -351,8 +351,8 @@ "value": "Limited independent assessment so far; recent GA release" } ], - "methodology": "Qualitative assessment; limited data given three weeks since GA", - "last_verified": "2026-06-10" + "methodology": "Qualitative assessment; limited data given ~7 weeks since GA", + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 90, @@ -366,7 +366,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 81, @@ -380,7 +380,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -394,10 +394,10 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Solid documentation at launch, but the model is only ~3 weeks GA; most benchmark figures remain vendor-reported and independent verification is still accumulating." + "notes": "Solid documentation at launch; ~7 weeks GA as of 2026-07-09. Most benchmark figures remain vendor-reported and independent verification is still accumulating." }, "operational_excellence": { @@ -415,7 +415,7 @@ } ], "methodology": "API design and feature completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -429,7 +429,7 @@ } ], "methodology": "SDK quality and maintenance assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 92, @@ -440,10 +440,16 @@ "url": "https://ai.google.dev/gemini-api/docs/changelog", "date": "2026-05-19", "value": "GA at I/O 2026 with documented model lifecycle" + }, + { + "source": "Gemini API Deprecations", + "url": "https://ai.google.dev/gemini-api/docs/deprecations", + "date": "2026-07-09", + "value": "gemini-3.5-flash active with no shutdown date; named designated replacement for gemini-2.0-flash (retired 2026-06-01), gemini-2.5-flash (shutdown 2026-10-16), and gemini-3-flash-preview" } ], "methodology": "Versioning and changelog review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 95, @@ -457,7 +463,7 @@ } ], "methodology": "Observability tooling review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 93, @@ -471,7 +477,7 @@ } ], "methodology": "Support channel assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 93, @@ -485,7 +491,7 @@ } ], "methodology": "Ecosystem and integration analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -499,7 +505,7 @@ } ], "methodology": "License terms review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Full Google Cloud operational stack from day one. Pricing ($1.50/$9.00, cached input $0.15) is notably higher than past Flash tiers, narrowing the cost gap to Pro models." @@ -555,9 +561,9 @@ "limitations": [ "Pricier than past Flash tiers ($1.50/$9.00 vs historical sub-dollar Flash pricing)", - "Benchmark figures are official claims; independent replication still accumulating (~3 weeks since GA)", + "Benchmark figures are official claims; independent replication still accumulating (~7 weeks since GA)", "Deepest reasoning tasks still favor Pro-tier models", - "Only ~3 weeks of production track record", + "Under two months of production track record", "Training data transparency limited (industry standard)" ], @@ -578,8 +584,8 @@ "pricing": { "input": "$1.50 per 1M tokens", "output": "$9.00 per 1M tokens", - "notes": "Cached input $0.15 per 1M. Pricier than past Flash tiers, reflecting 'frontier workhorse' positioning.", - "last_verified": "2026-06-10" + "notes": "Confirmed on the official Gemini API pricing page 2026-07-09. Cached input $0.15 per 1M; batch API 50% discount ($0.75/$4.50). Pricier than past Flash tiers, reflecting 'frontier workhorse' positioning.", + "last_verified": "2026-07-09" }, "context_window": 1048576, "max_output": 65536, diff --git a/data/models/gemini-3-flash.json b/data/models/gemini-3-flash.json index 2af24bb..bc0e6b7 100644 --- a/data/models/gemini-3-flash.json +++ b/data/models/gemini-3-flash.json @@ -4,9 +4,9 @@ "name": "Gemini 3 Flash", "provider": "Google", "version": "gemini-3-flash-preview", - "last_evaluated": "2026-01-14", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Google's efficiency model with Pro-level performance at 1/4 the price. 78% SWE-bench (beats Pro), 1M context, 3x faster than 2.5 Pro. Thinking level parameter for compute control.", + "description": "Google's efficiency model with Pro-level performance at low cost. 78% SWE-bench (beat Gemini 3 Pro), 1M context, 3x faster than 2.5 Pro. Thinking level parameter for compute control. Still served as gemini-3-flash-preview (no shutdown date announced), but superseded as Google's lead Flash tier by Gemini 3.5 Flash (2026-05-19), which Google lists as its designated replacement.", "website": "https://blog.google/products/gemini/gemini-3-flash/", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 94, @@ -45,7 +45,7 @@ } ], "methodology": "PhD-level reasoning benchmarks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 93, @@ -59,7 +59,7 @@ } ], "methodology": "Multimodal understanding testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 92, @@ -73,7 +73,7 @@ } ], "methodology": "Consistency testing across thinking levels", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.5s", @@ -87,7 +87,7 @@ } ], "methodology": "Median latency measurements", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "1.5s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile measurements", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -115,7 +115,7 @@ } ], "methodology": "Official specification", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -129,7 +129,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Exceptional value: 78% SWE-bench beats Pro at 1/4 the price. 3x faster than 2.5 Pro with 1M context. Beats Pro on tool use and MMMU." @@ -150,7 +150,7 @@ } ], "methodology": "OWASP LLM security testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 87, @@ -164,7 +164,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 84, @@ -178,7 +178,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_safety": { "score": 89, @@ -192,7 +192,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "api_security": { "score": 88, @@ -206,7 +206,7 @@ } ], "methodology": "API security review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strong security inherited from Gemini 3 family. Google Cloud infrastructure provides enterprise-grade protection." @@ -227,7 +227,7 @@ } ], "methodology": "Cloud infrastructure review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -241,7 +241,7 @@ } ], "methodology": "Terms review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Varies by tier", @@ -255,7 +255,7 @@ } ], "methodology": "Retention policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -269,7 +269,7 @@ } ], "methodology": "Data protection review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 90, @@ -283,7 +283,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 84, @@ -297,7 +297,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Good privacy with Google Cloud. Free tier available. Enterprise options for enhanced compliance." @@ -318,7 +318,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 87, @@ -332,7 +332,7 @@ } ], "methodology": "Factual accuracy testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -346,7 +346,7 @@ } ], "methodology": "Bias evaluation", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -360,7 +360,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 90, @@ -374,7 +374,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 82, @@ -388,7 +388,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -402,7 +402,7 @@ } ], "methodology": "Safety mechanism review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency with thinking level parameter. Configurable reasoning depth for different use cases." @@ -423,7 +423,7 @@ } ], "methodology": "API design review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -437,10 +437,10 @@ } ], "methodology": "SDK quality assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 90, + "score": 88, "confidence": "high", "evidence": [ { @@ -448,10 +448,16 @@ "url": "https://cloud.google.com/apis/design/versioning", "date": "2025-12-17", "value": "Clear versioning" + }, + { + "source": "Gemini API Deprecations", + "url": "https://ai.google.dev/gemini-api/docs/deprecations", + "date": "2026-07-09", + "value": "gemini-3-flash-preview listed with no announced shutdown date; designated replacement is gemini-3.5-flash" } ], "methodology": "Versioning policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -465,7 +471,7 @@ } ], "methodology": "Observability review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "support_quality": { "score": 91, @@ -479,7 +485,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 91, @@ -489,11 +495,11 @@ "source": "Google AI Ecosystem", "url": "https://ai.google.dev/", "date": "2025-12-17", - "value": "Default model in consumer Gemini app" + "value": "Launched as default model in consumer Gemini app (December 2025); Google's Flash positioning has since moved to Gemini 3.5 Flash" } ], "methodology": "Ecosystem analysis", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -507,10 +513,10 @@ } ], "methodology": "License review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, - "notes": "Excellent operational maturity. Default model in Gemini consumer app. Free tier available in API." + "notes": "Excellent operational maturity. Free tier available in API. Still in preview ~7 months after launch; Google's deprecations page names gemini-3.5-flash as its designated replacement, so plan migrations accordingly." } }, @@ -578,10 +584,10 @@ ], "limitations": [ - "Preview status (not yet GA)", + "Preview status (never promoted to GA; still gemini-3-flash-preview)", + "SUPERSEDED: Gemini 3.5 Flash (2026-05-19) is now Google's lead Flash tier and the designated replacement, though no shutdown date is announced", "Slightly behind Pro on GPQA Diamond (90.4% vs 93.8%)", "Less deep reasoning than Pro's Deep Think mode", - "Newer model with less enterprise testing", "Slightly higher than 2.5 Flash pricing ($0.50 vs $0.30)" ], @@ -601,10 +607,10 @@ "metadata": { "pricing": { - "input": "$0.50 per 1M tokens", + "input": "$0.50 per 1M tokens (text/image/video), $1.00 (audio)", "output": "$3.00 per 1M tokens", - "notes": "1/4 the price of Gemini 3 Pro. 10x cheaper than GPT-4o input. Free tier available.", - "last_verified": "2026-01-14" + "notes": "Confirmed on the official Gemini API pricing page 2026-07-09. 1/4 the price of Gemini 3.1 Pro and 1/3 of Gemini 3.5 Flash. Free tier available.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 64000, @@ -620,9 +626,10 @@ "knowledge_cutoff": "January 2025" }, - "related_entities": ["gemini-3-pro", "gemini-2-0-flash", "claude-sonnet-4-5", "gpt-4o"], + "related_entities": ["gemini-3-5-flash", "gemini-3-1-pro", "gemini-3-pro", "gemini-2-0-flash", "claude-sonnet-4-5"], "tags": [ + "superseded", "cost-effective", "fast", "1m-tokens", diff --git a/data/models/gemini-3-pro.json b/data/models/gemini-3-pro.json index 8a3121a..b0a5174 100644 --- a/data/models/gemini-3-pro.json +++ b/data/models/gemini-3-pro.json @@ -4,9 +4,9 @@ "name": "Gemini 3 Pro", "provider": "Google", "version": "gemini-3-pro-preview", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Former Google flagship, superseded by Gemini 3.1 Pro (GA 2026-02-19; ARC-AGI-2 77.1% vs 31.1%). Still served, with 1M token context, 1501 LMArena Elo (first model >1500), Deep Think mode for complex reasoning, and native multimodal. New projects should prefer Gemini 3.1 Pro.", + "description": "SHUT DOWN: Google retired Gemini 3 Pro Preview on 2026-03-09 (the gemini-pro-latest alias moved to 3.1 Pro on 2026-03-06); it is no longer served via the Gemini API or AI Studio. Historically a former Google flagship with 1M token context, 1501 LMArena Elo (first model >1500), Deep Think mode, and native multimodal. Migrate to Gemini 3.1 Pro (same pricing, ARC-AGI-2 77.1% vs 31.1%).", "website": "https://blog.google/products/gemini/gemini-3/", "trust_vector": { @@ -31,7 +31,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 97, @@ -63,7 +63,7 @@ } ], "methodology": "PhD-level and world-leading reasoning benchmarks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 96, @@ -89,7 +89,7 @@ } ], "methodology": "Crowdsourced and comprehensive testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 93, @@ -103,7 +103,7 @@ } ], "methodology": "Consistency and efficiency testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.5s", @@ -117,7 +117,7 @@ } ], "methodology": "Median latency measurements", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "4.0s", @@ -131,7 +131,7 @@ } ], "methodology": "95th percentile measurements", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -145,7 +145,7 @@ } ], "methodology": "Official specification", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -159,7 +159,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "First model to exceed 1500 LMArena Elo. 1M context enables unprecedented document processing. 6x improvement on ARC-AGI-2 over 2.5 Pro." @@ -180,7 +180,7 @@ } ], "methodology": "OWASP LLM security testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -194,7 +194,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 84, @@ -208,7 +208,7 @@ } ], "methodology": "Privacy policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_safety": { "score": 90, @@ -222,7 +222,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "api_security": { "score": 88, @@ -236,7 +236,7 @@ } ], "methodology": "API security review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strong security with Google Cloud infrastructure. Configurable safety filters provide flexibility." @@ -257,7 +257,7 @@ } ], "methodology": "Cloud infrastructure review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -271,7 +271,7 @@ } ], "methodology": "Terms review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Varies by tier", @@ -285,7 +285,7 @@ } ], "methodology": "Data retention policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 83, @@ -299,7 +299,7 @@ } ], "methodology": "Data protection review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 91, @@ -313,7 +313,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 84, @@ -327,7 +327,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Good privacy with Google Cloud. HIPAA compliance available through Google Cloud Healthcare API." @@ -348,7 +348,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -362,7 +362,7 @@ } ], "methodology": "Factual QA testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -376,7 +376,7 @@ } ], "methodology": "Bias benchmark evaluation", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 87, @@ -390,7 +390,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -404,7 +404,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 83, @@ -418,7 +418,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "guardrails": { "score": 91, @@ -432,14 +432,14 @@ } ], "methodology": "Safety mechanism review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency with Deep Think mode. Comprehensive documentation and configurable guardrails." }, "operational_excellence": { - "overall_score": 93, + "overall_score": 91, "criteria": { "api_design_quality": { "score": 95, @@ -453,7 +453,7 @@ } ], "methodology": "API design review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -467,10 +467,10 @@ } ], "methodology": "SDK quality assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 91, + "score": 80, "confidence": "high", "evidence": [ { @@ -480,14 +480,14 @@ "value": "Clear versioning with migration guides" }, { - "source": "Gemini API Changelog", - "url": "https://ai.google.dev/gemini-api/docs/changelog", - "date": "2026-06-10", - "value": "Superseded by Gemini 3.1 Pro (GA 2026-02-19; ARC-AGI-2 77.1% vs 31.1%)" + "source": "Gemini API Deprecations", + "url": "https://ai.google.dev/gemini-api/docs/deprecations", + "date": "2026-07-09", + "value": "gemini-3-pro-preview shut down 2026-03-09; gemini-pro-latest alias moved to 3.1 Pro on 2026-03-06; designated replacement is gemini-3.1-pro-preview (released 2026-02-19; ARC-AGI-2 77.1% vs 31.1%)" } ], "methodology": "Versioning policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 95, @@ -501,7 +501,7 @@ } ], "methodology": "Observability tools review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "support_quality": { "score": 93, @@ -515,7 +515,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 92, @@ -529,7 +529,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -543,10 +543,10 @@ } ], "methodology": "License review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, - "notes": "Excellent operational maturity with Google Cloud. First same-day launch across all Google AI platforms." + "notes": "Model retired 2026-03-09 after roughly four months in preview; no longer served. Operational history reflects its time in service on Google Cloud infrastructure." } }, @@ -619,7 +619,7 @@ "Deep Think increases latency significantly", "Data retention policies less clear than Anthropic", "Newer model with less community testing", - "SUPERSEDED: Gemini 3.1 Pro (GA 2026-02-19) is now Google's most capable Pro model; this model is no longer the flagship" + "SHUT DOWN 2026-03-09: gemini-3-pro-preview is no longer served; migrate to Gemini 3.1 Pro" ], "best_for": [ @@ -631,8 +631,8 @@ ], "not_recommended_for": [ + "Any use: the model was shut down 2026-03-09 and API calls fail", "Applications requiring lowest possible latency", - "Projects needing GA/stable model status", "Highly specialized coding (prefer Codex/Claude Opus)" ], @@ -641,8 +641,8 @@ "input": "$2.00 per 1M tokens (<200K), $4.00 per 1M tokens (>200K)", "output": "$12.00 per 1M tokens (<200K), $18.00 per 1M tokens (>200K)", "consumer": "$19.99/month (Google AI Pro), $124.99/month (Gemini 3 Ultra)", - "notes": "Tiered pricing based on context length", - "last_verified": "2026-01-14" + "notes": "Historical pricing; model shut down 2026-03-09. Gemini 3.1 Pro carries identical API pricing.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 64000, @@ -661,7 +661,7 @@ "related_entities": ["gemini-3-1-pro", "gemini-3-flash", "gemini-2-5-pro", "claude-opus-4-5", "gpt-5-2"], "tags": [ - "superseded", + "retired", "long-context", "1m-tokens", "deep-think", diff --git a/data/models/gemma-3-27b.json b/data/models/gemma-3-27b.json index 2a37418..2d15167 100644 --- a/data/models/gemma-3-27b.json +++ b/data/models/gemma-3-27b.json @@ -4,7 +4,7 @@ "name": "Gemma 3 27B", "provider": "Google", "version": "2025-01", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Google's open-source Gemma 3 model with 27 billion parameters, now superseded by Gemma 4 (released 2026-04-02 under Apache 2.0, a license improvement over the custom Gemma license). Was designed for developers seeking Google's research quality with open-source flexibility; new deployments should evaluate Gemma 4 instead.", "website": "https://ai.google.dev/gemma", @@ -24,7 +24,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 69, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 72, @@ -52,7 +52,7 @@ } ], "methodology": "Knowledge testing benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 71, @@ -66,7 +66,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.0s", @@ -80,7 +80,7 @@ } ], "methodology": "Median latency on recommended hardware", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.0s", @@ -94,21 +94,21 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { - "value": "8,192 tokens", + "value": "128,000 tokens", "confidence": "high", "evidence": [ { - "source": "Google Documentation", - "url": "https://ai.google.dev/gemma/docs/model_card_3", - "date": "2025-01-10", - "value": "8K token context window" + "source": "Gemma 3 Model Card", + "url": "https://ai.google.dev/gemma/docs/core/model_card_3", + "date": "2026-07-09", + "value": "128K token context window (all Gemma 3 sizes except 1B); prior 8K figure was a carryover from Gemma 2" } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 95, @@ -122,10 +122,10 @@ } ], "methodology": "User-controlled deployment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, - "notes": "Moderate performance suitable for basic tasks. Limited by smaller context window (8K tokens). Open-source flexibility." + "notes": "Moderate performance suitable for basic tasks. Context window corrected 2026-07-09 to 128K per the official Gemma 3 model card (the 8K figure previously listed was a Gemma 2 carryover). Open-source flexibility." }, "security": { "overall_score": 78, @@ -142,7 +142,7 @@ } ], "methodology": "Testing against prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 77, @@ -156,7 +156,7 @@ } ], "methodology": "Testing against adversarial prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -170,7 +170,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 78, @@ -184,7 +184,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 80, @@ -198,7 +198,7 @@ } ], "methodology": "Review of deployment practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Basic security with self-hosted deployment control. Additional safety layers recommended for production." @@ -218,7 +218,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 98, @@ -232,7 +232,7 @@ } ], "methodology": "Analysis of data flow", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "User-controlled", @@ -246,7 +246,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 92, @@ -260,7 +260,7 @@ } ], "methodology": "Review of deployment architecture", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 92, @@ -274,7 +274,7 @@ } ], "methodology": "Review of deployment options", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 98, @@ -288,7 +288,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent privacy with self-hosted deployment. Full control over all data aspects." @@ -308,7 +308,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 78, @@ -322,7 +322,7 @@ } ], "methodology": "Community evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 82, @@ -336,7 +336,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 81, @@ -350,7 +350,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 90, @@ -364,7 +364,7 @@ } ], "methodology": "Review of documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 88, @@ -378,7 +378,7 @@ } ], "methodology": "Review of technical documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 86, @@ -392,7 +392,7 @@ } ], "methodology": "Review of safety systems", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency as open-source model from Google. Comprehensive documentation." @@ -412,7 +412,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 85, @@ -426,7 +426,7 @@ } ], "methodology": "Review of SDKs", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 78, @@ -446,7 +446,7 @@ } ], "methodology": "Review of versioning", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Self-hosted weights remain usable, but the line has moved to Gemma 4" }, "monitoring_observability": { @@ -461,7 +461,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 82, @@ -475,7 +475,7 @@ } ], "methodology": "Assessment of support", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 84, @@ -489,7 +489,7 @@ } ], "methodology": "Analysis of ecosystem", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 92, @@ -503,7 +503,7 @@ } ], "methodology": "Review of license", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good operational maturity with Google's backing. Easier deployment than larger models." @@ -512,7 +512,7 @@ "use_case_ratings": { "code-generation": { "overall": 68, - "notes": "Basic coding capabilities. Limited context window (8K) restricts complex projects.", + "notes": "Basic coding capabilities. 128K context handles mid-size codebases, but accuracy limits complex projects.", "alternatives": [ "llama-4-scout", "gpt-4-1-mini" @@ -523,12 +523,12 @@ "notes": "Adequate for basic customer support with privacy benefits.", "alternatives": [ "gpt-4-1-mini", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "content-creation": { "overall": 74, - "notes": "Good for short-form content. Limited by 8K context window.", + "notes": "Good for short-form content; 128K context also permits long-form drafting.", "alternatives": [ "gpt-4-1", "llama-4-scout" @@ -584,7 +584,7 @@ }, "creative-writing": { "overall": 73, - "notes": "Adequate for short creative writing. Context limit restricts long-form content.", + "notes": "Adequate for creative writing; 128K context supports long-form, though quality trails frontier models.", "alternatives": [ "gpt-4-1", "llama-4-behemoth" @@ -600,8 +600,7 @@ "Cost-effective for basic tasks" ], "limitations": [ - "Limited accuracy (42.4% MMLU) compared to larger models", - "Small context window (8K tokens)", + "Limited accuracy compared to larger models", "Moderate coding capabilities", "Requires infrastructure for deployment", "Not suitable for complex or specialized tasks", @@ -617,7 +616,6 @@ ], "not_recommended_for": [ "Complex coding or software engineering", - "Long document processing (limited context)", "Advanced reasoning or mathematical tasks", "Production applications requiring highest accuracy", "Large-scale enterprise deployments", @@ -627,22 +625,17 @@ "pricing": { "input": "Self-hosted (infrastructure costs)", "output": "Self-hosted (infrastructure costs)", - "notes": "Open-source model. Typically $0.20-0.60 per 1M tokens with optimized deployment." + "notes": "Open-source model. Typically $0.20-0.60 per 1M tokens with optimized deployment.", + "last_verified": "2026-07-09" }, - "context_window": 8192, + "context_window": 128000, "languages": [ "English", - "Spanish", - "French", - "German", - "Italian", - "Portuguese", - "Japanese", - "Korean", - "Chinese" + "140+ languages" ], "modalities": [ - "text" + "text", + "image (input)" ], "api_endpoint": "Self-hosted", "open_source": true, diff --git a/data/models/gemma-4.json b/data/models/gemma-4.json index 89ff0ad..9cf01ae 100644 --- a/data/models/gemma-4.json +++ b/data/models/gemma-4.json @@ -4,7 +4,7 @@ "name": "Gemma 4", "provider": "Google", "version": "4.0", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Google's open-weight family released April 2026 under Apache 2.0 (a shift from the custom Gemma license). Spans E2B/E4B edge models with 128K context and native audio up to a 31B dense model with 256K context. The 31B scores ~1452 on LMArena, No. 3 among open models.", "website": "https://blog.google/innovation-and-ai/technology/developers-tools/gemma-4/", @@ -24,7 +24,7 @@ } ], "methodology": "Vendor-reported coding benchmarks compared against open-weight peer class", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 80, @@ -38,7 +38,7 @@ } ], "methodology": "Reasoning benchmark review from launch materials and open-model leaderboards", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 86, @@ -52,7 +52,7 @@ } ], "methodology": "Crowdsourced human preference rankings on LMArena", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 80, @@ -66,7 +66,7 @@ } ], "methodology": "Community reports across deployment stacks; high variance by quantization level", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "Deployment-dependent", @@ -80,7 +80,7 @@ } ], "methodology": "Self-hosted model; latency is a function of deployer infrastructure", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "256K tokens (31B/26B); 128K (E2B/E4B)", @@ -94,7 +94,7 @@ } ], "methodology": "Official specification from launch announcement", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "value": "Self-hosted (deployer-controlled)", @@ -108,7 +108,7 @@ } ], "methodology": "No single provider SLA; assessed as deployment-dependent", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strongest open-weight showing from Google to date: 31B at ~1452 LMArena (No. 3 open). MoE 26B-A4B offers near-dense quality at 4B active params. Performance below proprietary frontier but excellent per-parameter efficiency." @@ -128,7 +128,7 @@ } ], "methodology": "OWASP LLM01 assessment relative to model class; deployer must add input filtering", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 74, @@ -142,7 +142,7 @@ } ], "methodology": "Adversarial testing of instruction-tuned checkpoints; open weights inherently allow guardrail removal", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 88, @@ -156,7 +156,7 @@ } ], "methodology": "Architectural assessment: no third-party data flow when self-hosted", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 80, @@ -170,7 +170,7 @@ } ], "methodology": "Safety testing of released checkpoints and available companion classifiers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 75, @@ -184,7 +184,7 @@ } ], "methodology": "Assessment of typical self-hosted serving stacks vs managed alternatives", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Security profile is deployment-dependent: excellent data isolation when self-hosted, but guardrails are removable and there is no managed abuse filtering unless the deployer adds it (e.g., ShieldGemma, Vertex AI)." @@ -204,7 +204,7 @@ } ], "methodology": "Architectural assessment of self-hosted deployment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -218,7 +218,7 @@ } ], "methodology": "Architectural assessment: inference data never leaves deployer", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Deployer-controlled (zero by default)", @@ -232,7 +232,7 @@ } ], "methodology": "Architectural assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 85, @@ -246,7 +246,7 @@ } ], "methodology": "Data flow analysis for self-hosted inference", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 60, @@ -260,7 +260,7 @@ } ], "methodology": "Review of certification inheritance paths for open-weight deployments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -274,7 +274,7 @@ } ], "methodology": "Architectural assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Best-in-class data sovereignty: nothing leaves deployer infrastructure. The trade-off is that compliance certifications are not inherited from the model and must be built or bought by the deployer." @@ -294,7 +294,7 @@ } ], "methodology": "Assessment of inspection capabilities afforded by open weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 75, @@ -308,7 +308,7 @@ } ], "methodology": "Factual QA testing relative to model size class", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 78, @@ -322,7 +322,7 @@ } ], "methodology": "Model card review and independent audit availability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 76, @@ -336,7 +336,7 @@ } ], "methodology": "Calibration assessment; logprob access partially offsets weaker verbal uncertainty", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -350,7 +350,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -364,7 +364,7 @@ } ], "methodology": "Public disclosure review against open-model norms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 78, @@ -378,7 +378,7 @@ } ], "methodology": "Analysis of built-in and companion safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "High transparency by open-model standards: published technical report, architecture disclosure (including MoE active-parameter counts), and fully auditable weights. Apache 2.0 relicensing further reduces legal opacity." @@ -398,7 +398,7 @@ } ], "methodology": "Review of available serving interfaces and their consistency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 82, @@ -412,7 +412,7 @@ } ], "methodology": "Ecosystem tooling support assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 80, @@ -426,7 +426,7 @@ } ], "methodology": "Release cadence and immutability review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 65, @@ -440,7 +440,7 @@ } ], "methodology": "Assessment of out-of-box observability versus managed APIs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 70, @@ -454,7 +454,7 @@ } ], "methodology": "Support channel assessment for open-weight distribution", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 88, @@ -468,7 +468,7 @@ } ], "methodology": "Third-party integration and adoption analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 97, @@ -482,7 +482,7 @@ } ], "methodology": "License analysis; Apache 2.0 is OSI-approved with no usage restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Apache 2.0 relicensing is the headline trust improvement: prior Gemma generations carried custom-license use restrictions. Operational burden (monitoring, scaling, support) falls on the deployer, as with any open-weight model." @@ -572,7 +572,7 @@ "input": "Free (open weights; compute costs only)", "output": "Free (open weights; compute costs only)", "notes": "Apache 2.0. Self-hosting compute is the only cost; managed hosting available via Vertex AI and third-party providers.", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": 262144, "max_output": 32768, diff --git a/data/models/glm-5.json b/data/models/glm-5.json index f2eaf35..781fae1 100644 --- a/data/models/glm-5.json +++ b/data/models/glm-5.json @@ -4,7 +4,7 @@ "name": "GLM-5", "provider": "Z.ai (Zhipu AI)", "version": "20260211", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Z.ai's MIT-licensed 744B-parameter MoE (40B active) with 77.8% SWE-bench Verified, 92.7% AIME 2026, and open-source leadership on BrowseComp and agentic benchmarks. Trained on 28.5T tokens with DeepSeek Sparse Attention.", "website": "https://huggingface.co/zai-org/GLM-5", @@ -31,7 +31,7 @@ } ], "methodology": "Industry-standard coding benchmarks with independent third-party verification", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 93, @@ -51,7 +51,7 @@ } ], "methodology": "Graduate and competition-level reasoning benchmarks requiring multi-step problem solving", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 90, @@ -65,7 +65,7 @@ } ], "methodology": "Independent agentic and general-capability benchmarking across domains", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 87, @@ -79,7 +79,7 @@ } ], "methodology": "Community testing with repeated prompts and long agent runs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "2.6s", @@ -93,7 +93,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "6.0s", @@ -107,7 +107,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "200,000 tokens", @@ -121,7 +121,7 @@ } ], "methodology": "Official specification from model card", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 95, @@ -135,7 +135,7 @@ } ], "methodology": "Review of platform availability and self-hosting fallback options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Among the strongest open-weight models released to date: 77.8% SWE-bench Verified, 92.7% AIME 2026, 86.0% GPQA-Diamond, with independent confirmation of open-source leadership on BrowseComp, Vending Bench 2, and MCP-Atlas." @@ -156,7 +156,7 @@ } ], "methodology": "Review of safety documentation and community testing against OWASP LLM01 patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 78, @@ -170,7 +170,7 @@ } ], "methodology": "Testing against adversarial prompt datasets; deployer-dependent for self-hosted use", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 76, @@ -184,7 +184,7 @@ } ], "methodology": "Analysis of privacy policies and self-hosting data-control options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 81, @@ -198,7 +198,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 84, @@ -212,7 +212,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Solid for an open model; no published third-party security audit. Self-hosting shifts responsibility to the deployer." @@ -239,7 +239,7 @@ } ], "methodology": "Review of provider jurisdiction and third-party hosting options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 76, @@ -253,7 +253,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Per Z.ai policy on first-party API (China jurisdiction); zero when self-hosted", @@ -267,7 +267,7 @@ } ], "methodology": "Review of terms of service and deployment-dependent retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 72, @@ -281,7 +281,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 68, @@ -295,7 +295,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 85, @@ -309,7 +309,7 @@ } ], "methodology": "Review of self-hosting deployment options enabling zero retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "First-party Z.ai API operates under Chinese jurisdiction — a material caveat for Western regulated industries. The unencumbered MIT license makes self-hosting or Western-host deployment a clean mitigation." @@ -330,7 +330,7 @@ } ], "methodology": "Evaluation of reasoning transparency and trajectory inspectability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 81, @@ -344,7 +344,7 @@ } ], "methodology": "Testing on factual QA and grounded research benchmarks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 75, @@ -358,7 +358,7 @@ } ], "methodology": "Review of published bias benchmarks and community evaluations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 78, @@ -372,7 +372,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 89, @@ -386,7 +386,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 79, @@ -400,7 +400,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 78, @@ -414,7 +414,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Above-average transparency for an open frontier model: architecture, training scale, and benchmarks well documented with independent verification. Bias and safety evaluations remain thin." @@ -435,7 +435,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 82, @@ -449,7 +449,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 78, @@ -460,10 +460,16 @@ "url": "https://huggingface.co/zai-org", "date": "2026-04-07", "value": "Fast cadence: GLM-5.1 API launched 2026-03-27 with weights on 2026-04-07, same architecture; GLM-5 weights remain available" + }, + { + "source": "Z.ai release notes / GLM-5.2 coverage", + "url": "https://docs.z.ai/release-notes/new-released", + "date": "2026-07-09", + "value": "GLM-5.2 released mid-June 2026: MIT-licensed 744B/40B MoE with a 1M-token context window, priced at $1.40/$4.40 per 1M; GLM-5 remains served and its weights remain available" } ], "methodology": "Review of versioning practices and weight availability across releases", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 78, @@ -477,7 +483,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 79, @@ -491,7 +497,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 86, @@ -505,7 +511,7 @@ } ], "methodology": "Analysis of third-party hosting, integrations, and tooling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 95, @@ -519,10 +525,10 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Clean MIT licensing and strong ecosystem support. Note the rapid successor cadence: GLM-5.1 (same architecture) shipped within two months; evaluate which version your hosts actually serve." + "notes": "Clean MIT licensing and strong ecosystem support. Note the rapid successor cadence: GLM-5.1 shipped within two months and GLM-5.2 (1M context, MIT) within four; evaluate which version your hosts actually serve. GLM-5 weights remain available." } }, @@ -591,7 +597,7 @@ "limitations": [ "First-party Z.ai API processes data under Chinese jurisdiction with limited Western compliance certifications", "Text-only — no vision or audio modalities", - "Rapid successor cadence (GLM-5.1 within two months) creates version-tracking overhead", + "Rapid successor cadence (GLM-5.1 within two months, GLM-5.2 with 1M context by June 2026) creates version-tracking overhead", "Limited published bias, safety, and red-team evaluations", "Self-hosting a 744B MoE requires substantial GPU infrastructure", "English-language enterprise support is thin compared to Western providers" @@ -612,10 +618,10 @@ "metadata": { "pricing": { - "input": "$0.60 per 1M tokens (approx.)", - "output": "$1.92 per 1M tokens (approx.)", - "notes": "First-party Z.ai API pricing; third-party hosts vary. Successor GLM-5.1 priced similarly.", - "last_verified": "2026-06-10" + "input": "$0.60 per 1M tokens", + "output": "$1.92 per 1M tokens", + "notes": "First-party Z.ai API pricing, re-confirmed July 2026; third-party hosts vary. Successors are pricier: GLM-5.1 ~$0.97/$3.04 and GLM-5.2 $1.40/$4.40 per 1M, leaving GLM-5 as the budget option in the family.", + "last_verified": "2026-07-09" }, "context_window": 200000, "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], diff --git a/data/models/gpt-4-1-mini.json b/data/models/gpt-4-1-mini.json index d3115c5..45d5469 100644 --- a/data/models/gpt-4-1-mini.json +++ b/data/models/gpt-4-1-mini.json @@ -3,10 +3,10 @@ "type": "model", "name": "GPT-4.1 mini", "provider": "OpenAI", - "version": "2025-01", - "last_evaluated": "2025-11-08", + "version": "gpt-4.1-mini-2025-04-14", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's balanced GPT-4.1 variant offering good performance with efficient resource usage. Optimized for production workloads requiring quality outputs at reasonable cost.", + "description": "LEGACY: retired from ChatGPT 2026-02-13 but still available in the API with no announced shutdown (as of 2026-07-09). Balanced GPT-4.1 variant with a 1,047,576-token context window, offering good performance at reasonable cost. OpenAI recommends GPT-5.x mini tiers for new work.", "website": "https://openai.com/gpt-4-1", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 78, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 80, @@ -58,7 +58,7 @@ } ], "methodology": "Crowdsourced comparisons and knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 79, @@ -72,7 +72,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.8s", @@ -86,7 +86,7 @@ } ], "methodology": "Median latency for API requests", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "1.6s", @@ -100,21 +100,21 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { - "value": "128,000 tokens", + "value": "1,047,576 tokens", "confidence": "high", "evidence": [ { - "source": "OpenAI API Documentation", - "url": "https://platform.openai.com/docs/models/gpt-4-1-mini", - "date": "2025-01-15", - "value": "128K token context window" + "source": "OpenAI Model Page: gpt-4.1-mini", + "url": "https://developers.openai.com/api/docs/models/gpt-4.1-mini", + "date": "2026-07-09", + "value": "1,047,576 token context window; 32,768 max output tokens" } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -128,7 +128,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Balanced performance with good speed. Suitable for most production workloads requiring reliable outputs without premium pricing." @@ -148,7 +148,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 85, @@ -162,7 +162,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -176,7 +176,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 86, @@ -190,7 +190,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -204,7 +204,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong security posture with robust safety measures. Good balance of safety and usability." @@ -224,7 +224,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -238,7 +238,7 @@ } ], "methodology": "Analysis of privacy policy", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -252,7 +252,7 @@ } ], "methodology": "Review of terms of service", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -266,7 +266,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -280,7 +280,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -294,7 +294,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI privacy practices with SOC 2 compliance. 30-day retention period." @@ -314,7 +314,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -328,7 +328,7 @@ } ], "methodology": "Testing on factual QA datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 78, @@ -342,7 +342,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 79, @@ -356,7 +356,7 @@ } ], "methodology": "Qualitative assessment of confidence expression", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 85, @@ -370,7 +370,7 @@ } ], "methodology": "Review of documentation completeness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -384,7 +384,7 @@ } ], "methodology": "Review of public disclosures", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 84, @@ -398,7 +398,7 @@ } ], "methodology": "Analysis of safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with reasonable explainability. Moderate hallucination rate suitable for most applications." @@ -418,7 +418,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -432,7 +432,7 @@ } ], "methodology": "Review of SDK quality", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 85, @@ -443,10 +443,16 @@ "url": "https://platform.openai.com/docs/versioning", "date": "2025-01-15", "value": "Clear versioning policy" + }, + { + "source": "OpenAI: Retiring GPT-4o and older models", + "url": "https://openai.com/index/retiring-gpt-4o-and-older-models/", + "date": "2026-07-09", + "value": "GPT-4.1 mini retired from ChatGPT 2026-02-13; API access continues with no announced shutdown (not on the API deprecations list as of 2026-07-09)" } ], "methodology": "Review of versioning approach", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 84, @@ -460,7 +466,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 87, @@ -474,7 +480,7 @@ } ], "methodology": "Assessment of support channels", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 94, @@ -488,7 +494,7 @@ } ], "methodology": "Analysis of integrations", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -502,7 +508,7 @@ } ], "methodology": "Review of licensing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent operational maturity with OpenAI's established infrastructure and ecosystem." @@ -522,7 +528,7 @@ "notes": "Well-suited for customer support with fast response times and good conversational ability.", "alternatives": [ "gpt-4-1", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "content-creation": { @@ -592,7 +598,7 @@ "strengths": [ "Balanced performance and cost efficiency", "Fast response times (~0.8s p50) suitable for production", - "Large 128K context window for document processing", + "Very large context window (1,047,576 tokens) for document processing", "Good general knowledge (65% MMLU)", "Strong OpenAI ecosystem and tooling support", "Reliable uptime and infrastructure" @@ -603,7 +609,8 @@ "Not HIPAA eligible", "Moderate hallucination rate requires validation", "Limited regional data residency options", - "Not suitable for highly specialized or complex tasks" + "Not suitable for highly specialized or complex tasks", + "LEGACY: retired from ChatGPT 2026-02-13; API continues but OpenAI recommends GPT-5.x mini tiers for new work" ], "best_for": [ "Production applications requiring balanced quality and cost", @@ -622,11 +629,13 @@ ], "metadata": { "pricing": { - "input": "$0.60 per 1M tokens", - "output": "$1.80 per 1M tokens", - "notes": "Mid-tier pricing for balanced performance" + "input": "$0.40 per 1M tokens", + "output": "$1.60 per 1M tokens", + "notes": "Cached input $0.10 per 1M. Confirmed on official model page 2026-07-09; no longer listed on OpenAI's main pricing page.", + "last_verified": "2026-07-09" }, - "context_window": 128000, + "context_window": 1047576, + "max_output": 32768, "languages": [ "English", "Spanish", diff --git a/data/models/gpt-4-1-nano.json b/data/models/gpt-4-1-nano.json index e14d312..3d58b1c 100644 --- a/data/models/gpt-4-1-nano.json +++ b/data/models/gpt-4-1-nano.json @@ -3,10 +3,10 @@ "type": "model", "name": "GPT-4.1 nano", "provider": "OpenAI", - "version": "2025-01", - "last_evaluated": "2025-11-08", + "version": "gpt-4.1-nano-2025-04-14", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's smallest and most efficient GPT-4.1 variant, designed for high-volume, cost-sensitive applications. Optimized for speed and resource efficiency with basic capabilities.", + "description": "DEPRECATED: OpenAI announced 2026-04-22 that gpt-4.1-nano's API shuts down 2026-10-23; recommended replacement is gpt-5.4-nano. Historically OpenAI's smallest and most efficient GPT-4.1 variant for high-volume, cost-sensitive applications, with a 1,047,576-token context window.", "website": "https://openai.com/gpt-4-1", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring basic programming tasks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 66, @@ -38,7 +38,7 @@ } ], "methodology": "Basic reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 70, @@ -58,7 +58,7 @@ } ], "methodology": "Crowdsourced comparisons and knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 72, @@ -72,7 +72,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "More variance in outputs compared to larger models" }, "latency_p50": { @@ -87,7 +87,7 @@ } ], "methodology": "Median latency for API requests", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "0.8s", @@ -101,21 +101,21 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { - "value": "32,000 tokens", + "value": "1,047,576 tokens", "confidence": "high", "evidence": [ { - "source": "OpenAI API Documentation", - "url": "https://platform.openai.com/docs/models/gpt-4-1-nano", - "date": "2025-01-15", - "value": "32K token context window" + "source": "OpenAI Model Page: gpt-4.1-nano", + "url": "https://developers.openai.com/api/docs/models/gpt-4.1-nano", + "date": "2026-07-09", + "value": "1,047,576 token context window; 32,768 max output tokens" } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -129,7 +129,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Basic performance optimized for speed and efficiency. Best for simple tasks where ultra-low latency and cost are priorities." @@ -149,7 +149,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 80, @@ -163,7 +163,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -177,7 +177,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 84, @@ -191,7 +191,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -205,7 +205,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good security posture with standard OpenAI safety measures. Smaller model may have slightly lower resistance to adversarial attacks." @@ -225,7 +225,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -239,7 +239,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -253,7 +253,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -267,7 +267,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -281,7 +281,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -295,7 +295,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI privacy practices. 30-day data retention for abuse monitoring." @@ -315,7 +315,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 74, @@ -329,7 +329,7 @@ } ], "methodology": "Testing on factual QA datasets", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Higher hallucination rate than larger models" }, "bias_fairness": { @@ -344,7 +344,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 73, @@ -358,7 +358,7 @@ } ], "methodology": "Qualitative assessment of confidence expression", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 82, @@ -372,7 +372,7 @@ } ], "methodology": "Review of documentation completeness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -386,7 +386,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 80, @@ -400,7 +400,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Basic transparency features. Smaller model size limits explainability depth. Higher hallucination rate than premium models." @@ -420,7 +420,7 @@ } ], "methodology": "Review of API design and consistency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -434,10 +434,10 @@ } ], "methodology": "Review of SDK quality and maintenance", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 85, + "score": 78, "confidence": "high", "evidence": [ { @@ -445,10 +445,16 @@ "url": "https://platform.openai.com/docs/versioning", "date": "2025-01-15", "value": "Clear versioning with deprecation notices" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-07-09", + "value": "gpt-4.1-nano deprecation announced 2026-04-22; API shutdown 2026-10-23; recommended replacement gpt-5.4-nano" } ], "methodology": "Review of versioning policy", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 84, @@ -462,7 +468,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 87, @@ -476,7 +482,7 @@ } ], "methodology": "Assessment of support channels", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 94, @@ -490,7 +496,7 @@ } ], "methodology": "Analysis of third-party integrations", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -504,10 +510,10 @@ } ], "methodology": "Review of licensing terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, - "notes": "Excellent operational maturity leveraging OpenAI's established infrastructure. Same high-quality developer experience as larger models." + "notes": "Excellent operational maturity leveraging OpenAI's established infrastructure. Versioning score reduced to reflect deprecation: API shutdown 2026-10-23, migrate to gpt-5.4-nano." } }, "use_case_ratings": { @@ -516,7 +522,7 @@ "notes": "Basic code generation for simple tasks. 29.4% HumanEval indicates limited capability for complex programming.", "alternatives": [ "gpt-4-1", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "customer-support": { @@ -524,7 +530,7 @@ "notes": "Good for high-volume, simple customer queries. Fast response times make it suitable for basic support automation.", "alternatives": [ "gpt-4-1-mini", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "content-creation": { @@ -593,9 +599,9 @@ }, "strengths": [ "Ultra-low latency (~0.4s p50) ideal for real-time applications", - "Most cost-effective option in GPT-4.1 family", + "Most cost-effective option in GPT-4.1 family ($0.10/$0.40 per 1M)", "Good for high-volume, simple tasks", - "Smaller context window reduces processing overhead", + "Very large context window (1,047,576 tokens) at nano pricing", "Same API and ecosystem as premium OpenAI models", "Reliable uptime and infrastructure" ], @@ -605,7 +611,7 @@ "Higher hallucination rate than larger models", "Not suitable for complex or specialized tasks", "30-day data retention", - "Limited context window (32K tokens)" + "DEPRECATED: API shutdown 2026-10-23 (announced 2026-04-22); migrate to gpt-5.4-nano" ], "best_for": [ "High-volume, cost-sensitive applications", @@ -615,20 +621,22 @@ "Applications where speed is more important than accuracy" ], "not_recommended_for": [ + "New projects (API shutdown 2026-10-23; use gpt-5.4-nano instead)", "Complex coding or software development", "Advanced reasoning or mathematical tasks", "Healthcare or legal applications", "Tasks requiring deep domain expertise", - "Applications where accuracy is critical", - "Long document analysis (limited context window)" + "Applications where accuracy is critical" ], "metadata": { "pricing": { - "input": "$0.15 per 1M tokens", - "output": "$0.60 per 1M tokens", - "notes": "Most cost-effective option for high-volume applications" + "input": "$0.10 per 1M tokens", + "output": "$0.40 per 1M tokens", + "notes": "Cached input $0.025 per 1M. Confirmed on official model page 2026-07-09; model deprecated with API shutdown 2026-10-23.", + "last_verified": "2026-07-09" }, - "context_window": 32000, + "context_window": 1047576, + "max_output": 32768, "languages": [ "English", "Spanish", @@ -655,6 +663,7 @@ "gemma-3-27b" ], "tags": [ + "deprecated", "efficient", "low-latency", "cost-effective", diff --git a/data/models/gpt-4-1.json b/data/models/gpt-4-1.json index c5340d3..07c4a61 100644 --- a/data/models/gpt-4-1.json +++ b/data/models/gpt-4-1.json @@ -3,10 +3,10 @@ "type": "model", "name": "GPT-4.1", "provider": "OpenAI", - "version": "2025-01", - "last_evaluated": "2025-11-08", + "version": "gpt-4.1-2025-04-14", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's flagship GPT-4.1 model offering strong general-purpose capabilities across diverse tasks. The standard choice for production applications requiring reliable, high-quality outputs.", + "description": "LEGACY: retired from ChatGPT 2026-02-13 but still available in the API with no announced shutdown (as of 2026-07-09). Previous-generation general-purpose GPT-4.1 model with a 1,047,576-token context window. OpenAI recommends GPT-5.x models for new work.", "website": "https://openai.com/gpt-4-1", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 84, @@ -50,7 +50,7 @@ } ], "methodology": "Mathematical and scientific reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 86, @@ -70,7 +70,7 @@ } ], "methodology": "Crowdsourced comparisons and comprehensive knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 85, @@ -84,7 +84,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.2s", @@ -98,7 +98,7 @@ } ], "methodology": "Median latency for API requests", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.4s", @@ -112,21 +112,21 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { - "value": "128,000 tokens", + "value": "1,047,576 tokens", "confidence": "high", "evidence": [ { - "source": "OpenAI API Documentation", - "url": "https://platform.openai.com/docs/models/gpt-4-1", - "date": "2025-01-15", - "value": "128K token context window" + "source": "OpenAI Model Page: gpt-4.1", + "url": "https://developers.openai.com/api/docs/models/gpt-4.1", + "date": "2026-07-09", + "value": "1,047,576 token context window; 32,768 max output tokens" } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -140,7 +140,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong general-purpose performance with good balance across coding, reasoning, and knowledge tasks. Flagship model for most production use cases." @@ -160,7 +160,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -174,7 +174,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -188,7 +188,7 @@ } ], "methodology": "Analysis of privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 87, @@ -202,7 +202,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -216,7 +216,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong security posture with comprehensive safety systems. Robust protection against adversarial attacks." @@ -236,7 +236,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -250,7 +250,7 @@ } ], "methodology": "Analysis of privacy policy", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -264,7 +264,7 @@ } ], "methodology": "Review of terms of service", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -278,7 +278,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -292,7 +292,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -306,7 +306,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Standard enterprise privacy practices with SOC 2 Type II certification. 30-day retention period." @@ -326,7 +326,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 82, @@ -340,7 +340,7 @@ } ], "methodology": "Testing on factual QA datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 79, @@ -354,7 +354,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 81, @@ -368,7 +368,7 @@ } ], "methodology": "Qualitative assessment of confidence expression", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 87, @@ -382,7 +382,7 @@ } ], "methodology": "Review of documentation completeness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -396,7 +396,7 @@ } ], "methodology": "Review of public disclosures", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 86, @@ -410,7 +410,7 @@ } ], "methodology": "Analysis of safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with solid explainability. Lower hallucination rate than smaller models. Comprehensive safety systems." @@ -430,7 +430,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -444,7 +444,7 @@ } ], "methodology": "Review of SDK quality", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 86, @@ -455,10 +455,16 @@ "url": "https://platform.openai.com/docs/versioning", "date": "2025-01-15", "value": "Clear versioning with deprecation notices" + }, + { + "source": "OpenAI: Retiring GPT-4o and older models", + "url": "https://openai.com/index/retiring-gpt-4o-and-older-models/", + "date": "2026-07-09", + "value": "GPT-4.1 retired from ChatGPT 2026-02-13; API access continues with no announced shutdown (not on the API deprecations list as of 2026-07-09)" } ], "methodology": "Review of versioning approach", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 85, @@ -472,7 +478,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 89, @@ -486,7 +492,7 @@ } ], "methodology": "Assessment of support channels", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 95, @@ -500,7 +506,7 @@ } ], "methodology": "Analysis of integrations", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -514,7 +520,7 @@ } ], "methodology": "Review of licensing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent operational maturity with industry-leading ecosystem and developer experience." @@ -603,7 +609,7 @@ "strengths": [ "Strong general-purpose performance (66.3% MMLU)", "Good balance of quality and speed (~1.2s p50)", - "Large 128K context window for document processing", + "Very large context window (1,047,576 tokens) for document processing", "Mature ecosystem with extensive integrations", "Reliable uptime and infrastructure (99.9%)", "Comprehensive safety and security features" @@ -614,7 +620,8 @@ "Not HIPAA eligible", "Limited regional data residency options", "Higher pricing than smaller models", - "Training data transparency limited" + "Training data transparency limited", + "LEGACY: retired from ChatGPT 2026-02-13; API continues but OpenAI recommends GPT-5.x for new work" ], "best_for": [ "General-purpose production applications", @@ -633,11 +640,13 @@ ], "metadata": { "pricing": { - "input": "$2.50 per 1M tokens", - "output": "$10.00 per 1M tokens", - "notes": "Standard flagship pricing" + "input": "$2.00 per 1M tokens", + "output": "$8.00 per 1M tokens", + "notes": "Cached input $0.50 per 1M. Confirmed on official model page 2026-07-09; no longer listed on OpenAI's main pricing page.", + "last_verified": "2026-07-09" }, - "context_window": 128000, + "context_window": 1047576, + "max_output": 32768, "languages": [ "English", "Spanish", diff --git a/data/models/gpt-4o-mini.json b/data/models/gpt-4o-mini.json index 17a05bc..e0fbea7 100644 --- a/data/models/gpt-4o-mini.json +++ b/data/models/gpt-4o-mini.json @@ -4,9 +4,9 @@ "name": "GPT-4o mini", "provider": "OpenAI", "version": "2024-07", - "last_evaluated": "2025-11-08", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's efficient multimodal model combining text and vision capabilities at competitive pricing. Designed for cost-sensitive applications requiring basic multimodal understanding.", + "description": "OpenAI's efficient multimodal model combining text and vision capabilities at competitive pricing. Designed for cost-sensitive applications requiring basic multimodal understanding. Legacy model: remains available in the API with no announced shutdown as of 2026-07-09 (unaffected by the GPT-4o retirement), but OpenAI recommends newer GPT-5.x mini/nano tiers for new work.", "website": "https://openai.com/gpt-4o", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 70, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 72, @@ -52,7 +52,7 @@ } ], "methodology": "Knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 73, @@ -66,7 +66,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.7s", @@ -80,7 +80,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "1.4s", @@ -94,7 +94,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -108,7 +108,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -122,7 +122,7 @@ } ], "methodology": "Historical uptime", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Basic multimodal performance with fast inference. Good for simple vision + text tasks." @@ -142,7 +142,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 83, @@ -156,7 +156,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -170,7 +170,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 84, @@ -184,7 +184,7 @@ } ], "methodology": "Safety benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -198,7 +198,7 @@ } ], "methodology": "Security review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good security with multimodal safety considerations." @@ -218,7 +218,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -232,7 +232,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -246,7 +246,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -260,7 +260,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -274,7 +274,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -288,7 +288,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI privacy with 30-day retention." @@ -308,7 +308,7 @@ } ], "methodology": "Reasoning evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 78, @@ -322,7 +322,7 @@ } ], "methodology": "Factual QA testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 79, @@ -336,7 +336,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 80, @@ -350,7 +350,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 85, @@ -364,7 +364,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -378,7 +378,7 @@ } ], "methodology": "Public disclosure", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 84, @@ -392,7 +392,7 @@ } ], "methodology": "Safety system analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with multimodal safety considerations." @@ -412,7 +412,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -426,7 +426,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 85, @@ -437,10 +437,16 @@ "url": "https://platform.openai.com/docs/versioning", "date": "2024-07-15", "value": "Clear versioning" + }, + { + "source": "OpenAI Model Page: gpt-4o-mini", + "url": "https://developers.openai.com/api/docs/models/gpt-4o-mini", + "date": "2026-07-09", + "value": "Model remains active in the API with no deprecation or shutdown date announced; not included in the 2026-02-13 ChatGPT retirements or the API deprecations list" } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 84, @@ -454,7 +460,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 87, @@ -468,7 +474,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 94, @@ -482,7 +488,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -496,7 +502,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent operational maturity with multimodal capabilities." @@ -515,7 +521,7 @@ "notes": "Good for support with vision (image understanding).", "alternatives": [ "gpt-4-1-mini", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "content-creation": { @@ -610,10 +616,11 @@ "pricing": { "input": "$0.15 per 1M tokens", "output": "$0.60 per 1M tokens", - "notes": "Cost-effective multimodal pricing", - "last_verified": "2025-11-09" + "notes": "Cost-effective multimodal pricing. Confirmed unchanged on official model page 2026-07-09; no longer listed on OpenAI's main pricing page.", + "last_verified": "2026-07-09" }, "context_window": 128000, + "max_output": 16384, "languages": [ "English", "Spanish", diff --git a/data/models/gpt-4o.json b/data/models/gpt-4o.json index 57432eb..859595b 100644 --- a/data/models/gpt-4o.json +++ b/data/models/gpt-4o.json @@ -4,9 +4,9 @@ "name": "GPT-4o", "provider": "OpenAI", "version": "2024-05", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DEPRECATED: removed from ChatGPT 2026-02-13; the gpt-4o-2024-05-13 snapshot's API shuts down 2026-10-23. Historically OpenAI's flagship multimodal model with strong text and vision capabilities for high-quality multimodal understanding and generation. Migrate to newer GPT-5.x models.", + "description": "DEPRECATED: removed from ChatGPT 2026-02-13 and fully retired from ChatGPT (including Custom GPTs) 2026-04-03; chatgpt-4o-latest API access ended 2026-02-16; the gpt-4o-2024-05-13 snapshot's API shuts down 2026-10-23 (gpt-4o-2024-11-20 remains served via API). Historically OpenAI's flagship multimodal model with strong text and vision capabilities for high-quality multimodal understanding and generation. Migrate to newer GPT-5.x models.", "website": "https://openai.com/gpt-4o", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 80, @@ -39,7 +39,7 @@ } ], "methodology": "Mathematical benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 82, @@ -59,7 +59,7 @@ } ], "methodology": "Knowledge testing and multimodal benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 83, @@ -73,7 +73,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.3s", @@ -87,7 +87,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.6s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -115,7 +115,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -129,7 +129,7 @@ } ], "methodology": "Historical uptime", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong multimodal performance with good balance of text and vision capabilities. Better general knowledge (56.1% MMLU) than mini variant." @@ -150,7 +150,7 @@ } ], "methodology": "Multimodal adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 86, @@ -164,7 +164,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -178,7 +178,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 86, @@ -192,7 +192,7 @@ } ], "methodology": "Safety benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -206,7 +206,7 @@ } ], "methodology": "Security review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong security with multimodal safety considerations. Good resistance to adversarial attacks across modalities." @@ -227,7 +227,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -241,7 +241,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -255,7 +255,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -269,7 +269,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Extra care needed with PII in images" }, "compliance_certifications": { @@ -284,7 +284,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -298,7 +298,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI privacy with 30-day retention. Extra considerations for image data." @@ -319,7 +319,7 @@ } ], "methodology": "Reasoning evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -333,7 +333,7 @@ } ], "methodology": "Factual QA testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 79, @@ -347,7 +347,7 @@ } ], "methodology": "Multimodal bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 81, @@ -361,7 +361,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 86, @@ -375,7 +375,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -389,7 +389,7 @@ } ], "methodology": "Public disclosure", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 86, @@ -403,7 +403,7 @@ } ], "methodology": "Safety system analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with comprehensive multimodal documentation. Strong safety guardrails across modalities." @@ -424,7 +424,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -438,7 +438,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 74, @@ -455,10 +455,16 @@ "url": "https://openai.com/index/retiring-gpt-4o-and-older-models/", "date": "2026-06-10", "value": "GPT-4o removed from ChatGPT 2026-02-13; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23" + }, + { + "source": "OpenAI Help Center: Retiring GPT-4o and other ChatGPT models", + "url": "https://help.openai.com/en/articles/20001051-retiring-gpt-4o-and-other-chatgpt-models", + "date": "2026-07-09", + "value": "GPT-4o fully retired from ChatGPT (all plans, incl. Custom GPTs) since 2026-04-03; chatgpt-4o-latest API access ended 2026-02-16; gpt-4o-2024-11-20 snapshot still served via API" } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 85, @@ -472,7 +478,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -486,7 +492,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 86, @@ -500,7 +506,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -514,10 +520,10 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, - "notes": "Deprecated: removed from ChatGPT 2026-02-13; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23. Versioning and ecosystem scores reduced to reflect deprecation." + "notes": "Deprecated: fully retired from ChatGPT since 2026-04-03; chatgpt-4o-latest API ended 2026-02-16; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23 (gpt-4o-2024-11-20 still served). Versioning and ecosystem scores reduced to reflect deprecation." } }, @@ -590,7 +596,7 @@ "Higher cost than text-only alternatives", "PII concerns with image inputs", "Moderate coding capabilities", - "DEPRECATED: removed from ChatGPT 2026-02-13; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23" + "DEPRECATED: fully retired from ChatGPT since 2026-04-03; chatgpt-4o-latest API ended 2026-02-16; gpt-4o-2024-05-13 snapshot API shutdown 2026-10-23" ], "best_for": [ @@ -614,10 +620,11 @@ "pricing": { "input": "$2.50 per 1M tokens", "output": "$10.00 per 1M tokens", - "notes": "Flagship multimodal pricing", - "last_verified": "2025-11-09" + "notes": "Legacy pricing unchanged (confirmed on official model page 2026-07-09; cached input $1.25). No longer listed on OpenAI's main pricing page.", + "last_verified": "2026-07-09" }, "context_window": 128000, + "max_output": 16384, "languages": [ "English", "Spanish", diff --git a/data/models/gpt-5-1.json b/data/models/gpt-5-1.json index 823b4a4..a84a97c 100644 --- a/data/models/gpt-5-1.json +++ b/data/models/gpt-5-1.json @@ -3,10 +3,10 @@ "type": "model", "name": "GPT-5.1", "provider": "OpenAI", - "version": "gpt-5-1-1113", - "last_evaluated": "2026-06-10", + "version": "gpt-5.1-2025-11-13", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "SUPERSEDED by GPT-5.2/5.4/5.5 (GPT-5.5 is the current flagship); gpt-5.1-chat-latest and gpt-5.1-codex variants shut down in the API 2026-07-23. Released Nov 2025 with adaptive reasoning (2-3x faster on simple tasks), 76.3% SWE-bench, developer tools (apply_patch, shell), warmer tone.", + "description": "SUPERSEDED by GPT-5.2/5.4/5.5 and the GPT-5.6 family (released 2026-07-09); retired from ChatGPT 2026-03-11. gpt-5.1-chat-latest, gpt-5.1-codex, gpt-5.1-codex-max, and gpt-5.1-codex-mini shut down in the API 2026-07-23 (migrate to gpt-5.5 / gpt-5.4-mini); the base gpt-5.1 snapshot remains served with no announced shutdown. Released Nov 2025 with adaptive reasoning (2-3x faster on simple tasks), 76.3% SWE-bench, developer tools (apply_patch, shell), warmer tone.", "website": "https://openai.com/index/gpt-5-1/", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Standard coding benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -45,7 +45,7 @@ } ], "methodology": "Graduate and PhD-level reasoning benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 96, @@ -59,7 +59,7 @@ } ], "methodology": "Crowdsourced blind comparisons", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 94, @@ -73,7 +73,7 @@ } ], "methodology": "Internal testing across temperature settings", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.6s", @@ -87,7 +87,7 @@ } ], "methodology": "Platform-wide performance metrics", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.8s", @@ -101,21 +101,21 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "context_window": { - "value": "128,000 tokens", + "value": "400,000 tokens", "confidence": "high", "evidence": [ { - "source": "OpenAI Documentation", - "url": "https://platform.openai.com/docs/models/gpt-5", - "date": "2025-01-01", - "value": "128K token context window" + "source": "OpenAI Model Page: gpt-5.1", + "url": "https://developers.openai.com/api/docs/models/gpt-5.1", + "date": "2026-07-09", + "value": "400K token context window; 128K max output tokens" } ], "methodology": "Official specification", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -129,7 +129,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "76.3% SWE-bench (best). Adaptive reasoning: 2-3x faster on simple tasks. New developer tools (apply_patch, shell). Warmer conversational tone." @@ -150,7 +150,7 @@ } ], "methodology": "Testing against OWASP LLM01 attacks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -164,7 +164,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 82, @@ -178,7 +178,7 @@ } ], "methodology": "Policy review and data handling practices", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_safety": { "score": 90, @@ -192,7 +192,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -206,7 +206,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Strong security with improved jailbreak resistance. Multi-layered safety systems provide robust output filtering." @@ -227,7 +227,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -241,7 +241,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -255,7 +255,7 @@ } ], "methodology": "Terms of service review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Zero retention available for enterprise customers" }, "pii_handling": { @@ -270,7 +270,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -284,7 +284,7 @@ } ], "methodology": "Verification of certifications", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Not HIPAA eligible currently" }, "zero_data_retention": { @@ -299,7 +299,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Good privacy posture with strong enterprise controls. 30-day default retention (vs Anthropic's 0-day). Not HIPAA eligible." @@ -320,7 +320,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -334,7 +334,7 @@ } ], "methodology": "Factual accuracy testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -348,7 +348,7 @@ } ], "methodology": "Bias benchmarks and demographic testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 87, @@ -362,7 +362,7 @@ } ], "methodology": "Qualitative confidence expression", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -376,7 +376,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -390,7 +390,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "guardrails": { "score": 91, @@ -404,7 +404,7 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Excellent transparency with unified thinking feature and comprehensive system card. Industry-leading hallucination prevention." @@ -425,7 +425,7 @@ } ], "methodology": "API design and feature review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -439,7 +439,7 @@ } ], "methodology": "SDK quality and maintenance review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 90, @@ -454,12 +454,18 @@ { "source": "OpenAI Deprecations", "url": "https://developers.openai.com/api/docs/deprecations", - "date": "2026-06-10", - "value": "gpt-5.1-chat-latest and gpt-5.1-codex variants API shutdown 2026-07-23; superseded by GPT-5.4/5.5" + "date": "2026-07-09", + "value": "gpt-5.1-chat-latest, gpt-5.1-codex, gpt-5.1-codex-max, gpt-5.1-codex-mini API shutdown 2026-07-23 (replacements gpt-5.5 / gpt-5.4-mini); base gpt-5.1 snapshot not on the deprecations list" + }, + { + "source": "OpenAI Help Center: Model Release Notes", + "url": "https://help.openai.com/en/articles/9624314-model-release-notes", + "date": "2026-07-09", + "value": "GPT-5.1 models retired from ChatGPT 2026-03-11 (chats and GPTs); still available via the API" } ], "methodology": "Versioning policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 92, @@ -473,7 +479,7 @@ } ], "methodology": "Observability tools review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "support_quality": { "score": 93, @@ -487,7 +493,7 @@ } ], "methodology": "Support and documentation assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 96, @@ -501,7 +507,7 @@ } ], "methodology": "Ecosystem breadth and depth analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -515,7 +521,7 @@ } ], "methodology": "License terms review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Industry-leading operational maturity with the most mature ecosystem. Excellent APIs, SDKs, and tooling." @@ -587,10 +593,9 @@ "limitations": [ "Not HIPAA eligible (unlike Claude models)", "30-day data retention vs Anthropic's 0-day default", - "Smaller context window (128K vs Claude's 200K)", "Premium pricing comparable to Claude", "Slightly behind Claude on specialized coding benchmarks", - "SUPERSEDED by GPT-5.4/5.5; gpt-5.1-chat-latest and gpt-5.1-codex variants API shutdown 2026-07-23" + "SUPERSEDED by GPT-5.4/5.5/5.6; retired from ChatGPT 2026-03-11; gpt-5.1-chat-latest and all gpt-5.1-codex variants API shutdown 2026-07-23" ], "best_for": [ @@ -610,12 +615,13 @@ "metadata": { "pricing": { - "input": "$2.50 per 1M tokens", - "output": "$20.00 per 1M tokens", - "notes": "Same as GPT-5. Batch API offers 50% discount. Faster = better value.", - "last_verified": "2025-11-17" + "input": "$1.25 per 1M tokens", + "output": "$10.00 per 1M tokens", + "notes": "Same as GPT-5; cached input $0.125 per 1M. Confirmed on official model page 2026-07-09; chat/codex variants shut down 2026-07-23.", + "last_verified": "2026-07-09" }, - "context_window": 128000, + "context_window": 400000, + "max_output": 128000, "languages": [ "English", "Spanish", diff --git a/data/models/gpt-5-2-codex.json b/data/models/gpt-5-2-codex.json index 29fce21..bfc4978 100644 --- a/data/models/gpt-5-2-codex.json +++ b/data/models/gpt-5-2-codex.json @@ -4,9 +4,9 @@ "name": "GPT-5.2 Codex", "provider": "OpenAI", "version": "gpt-5-2-codex-2025-12-11", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DEPRECATED: superseded by GPT-5.3-Codex (2026-02-05); gpt-5.2-codex API shuts down 2026-07-23. Historically OpenAI's specialized coding model built on GPT-5.2 with 56.4% SWE-bench Pro, 64% Terminal-bench 2.0, native code compaction, and enhanced cybersecurity capabilities.", + "description": "DEPRECATED: superseded by GPT-5.3-Codex (2026-02-05); gpt-5.2-codex API shuts down 2026-07-23 (two weeks from 2026-07-09 — migrate now; OpenAI's listed replacement is gpt-5.5). Historically OpenAI's specialized coding model built on GPT-5.2 with 56.4% SWE-bench Pro, 64% Terminal-bench 2.0, native code compaction, and enhanced cybersecurity capabilities.", "website": "https://openai.com/index/introducing-gpt-5-2-codex/", "trust_vector": { @@ -37,7 +37,7 @@ } ], "methodology": "Professional and enterprise coding benchmarks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 92, @@ -51,7 +51,7 @@ } ], "methodology": "Reasoning benchmarks optimized for code-related tasks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 88, @@ -65,7 +65,7 @@ } ], "methodology": "General knowledge testing", - "last_verified": "2026-01-14", + "last_verified": "2026-07-09", "notes": "Optimized for coding tasks; general performance slightly reduced" }, "output_consistency": { @@ -80,7 +80,7 @@ } ], "methodology": "Code consistency and format testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.0s", @@ -94,7 +94,7 @@ } ], "methodology": "Median latency for code generation", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.0s", @@ -108,7 +108,7 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "context_window": { "value": "400,000 tokens", @@ -122,7 +122,7 @@ } ], "methodology": "Official specification", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -136,7 +136,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "State-of-the-art coding model: 56.4% SWE-bench Pro, 82.1% SWE-bench Verified, 64% Terminal-bench 2.0. Native code compaction for clean outputs." @@ -157,7 +157,7 @@ } ], "methodology": "Testing against code-focused injection attacks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 90, @@ -171,7 +171,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 86, @@ -185,7 +185,7 @@ } ], "methodology": "Code-specific data handling review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_safety": { "score": 92, @@ -199,7 +199,7 @@ } ], "methodology": "Security-focused code output testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "api_security": { "score": 86, @@ -213,7 +213,7 @@ } ], "methodology": "API security review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Enhanced cybersecurity capabilities for secure code generation. Specialized for identifying and avoiding code vulnerabilities." @@ -234,7 +234,7 @@ } ], "methodology": "Enterprise documentation review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -248,7 +248,7 @@ } ], "methodology": "Policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -262,7 +262,7 @@ } ], "methodology": "Terms review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -276,7 +276,7 @@ } ], "methodology": "Data protection review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -290,7 +290,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 86, @@ -304,7 +304,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI privacy. Important for code: ensure proprietary code handling policies are understood." @@ -325,7 +325,7 @@ } ], "methodology": "Code explainability assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 90, @@ -339,7 +339,7 @@ } ], "methodology": "Code accuracy and compilation testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 83, @@ -353,7 +353,7 @@ } ], "methodology": "Code generation bias assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -367,7 +367,7 @@ } ], "methodology": "Code confidence expression", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 91, @@ -381,7 +381,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -395,7 +395,7 @@ } ], "methodology": "Training data disclosure review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "guardrails": { "score": 92, @@ -409,7 +409,7 @@ } ], "methodology": "Code safety mechanism review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strong code explainability with native documentation generation. Enhanced for secure code practices." @@ -430,7 +430,7 @@ } ], "methodology": "API design review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 95, @@ -444,7 +444,7 @@ } ], "methodology": "SDK review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 79, @@ -459,12 +459,12 @@ { "source": "OpenAI Deprecations", "url": "https://developers.openai.com/api/docs/deprecations", - "date": "2026-06-10", - "value": "gpt-5.2-codex API shutdown 2026-07-23; superseded by GPT-5.3-Codex (2026-02-05)" + "date": "2026-07-09", + "value": "Re-verified 2026-07-09: gpt-5.2-codex API shutdown 2026-07-23 (announced 2026-04-22); OpenAI's listed replacement is gpt-5.5; GPT-5.3-Codex is the direct Codex successor" } ], "methodology": "Versioning review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 93, @@ -478,7 +478,7 @@ } ], "methodology": "Observability review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "support_quality": { "score": 94, @@ -492,7 +492,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 88, @@ -506,7 +506,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -520,7 +520,7 @@ } ], "methodology": "License review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Deprecated: gpt-5.2-codex API shutdown scheduled 2026-07-23; superseded by GPT-5.3-Codex. Versioning and ecosystem scores reduced to reflect deprecation." @@ -620,7 +620,7 @@ "input": "$1.75 per 1M tokens", "output": "$14.00 per 1M tokens", "notes": "Same as GPT-5.2. Optimized for coding efficiency.", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "context_window": 400000, "max_output": 128000, diff --git a/data/models/gpt-5-2.json b/data/models/gpt-5-2.json index 9da8c08..e6bdabb 100644 --- a/data/models/gpt-5-2.json +++ b/data/models/gpt-5-2.json @@ -4,9 +4,9 @@ "name": "GPT-5.2", "provider": "OpenAI", "version": "gpt-5-2-2025-12-11", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "SUPERSEDED by GPT-5.4 (2026-03-05) and GPT-5.5 (2026-04-23, current OpenAI flagship). Still served, with 400K context window, 100% AIME 2025 score, and 52.9% ARC-AGI-2. Three variants: Instant (speed), Thinking (reasoning), Pro (accuracy). New projects should prefer GPT-5.5.", + "description": "SUPERSEDED by GPT-5.4 (2026-03-05), GPT-5.5 (2026-04-23), and the GPT-5.6 family (2026-07-09). API snapshots (gpt-5.2, gpt-5.2-2025-12-11) still served with no announced shutdown, but gpt-5.2-chat-latest shuts down 2026-08-10 (announced 2026-05-08; migrate to gpt-5.5). 400K context window, 100% AIME 2025 score, 52.9% ARC-AGI-2. Three variants: Instant (speed), Thinking (reasoning), Pro (accuracy). New projects should prefer GPT-5.5.", "website": "https://openai.com/index/introducing-gpt-5-2/", "trust_vector": { @@ -37,7 +37,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 98, @@ -75,7 +75,7 @@ } ], "methodology": "PhD-level and Olympiad-level reasoning benchmarks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 97, @@ -95,7 +95,7 @@ } ], "methodology": "Crowdsourced and expert-level comparisons", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 95, @@ -109,7 +109,7 @@ } ], "methodology": "Internal testing across model variants", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.5s (Instant) / 1.2s (Thinking)", @@ -123,7 +123,7 @@ } ], "methodology": "Platform-wide performance metrics", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.5s (standard)", @@ -137,7 +137,7 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "context_window": { "value": "400,000 tokens", @@ -151,7 +151,7 @@ } ], "methodology": "Official specification", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -165,7 +165,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Industry-leading reasoning with 100% AIME and 52.9% ARC-AGI-2. 400K context enables full codebase processing. ~30% fewer hallucinations than GPT-5.1." @@ -186,7 +186,7 @@ } ], "methodology": "Testing against OWASP LLM01 attacks", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 90, @@ -200,7 +200,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 84, @@ -214,7 +214,7 @@ } ], "methodology": "Policy review and data handling practices", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "output_safety": { "score": 91, @@ -228,7 +228,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "api_security": { "score": 86, @@ -242,7 +242,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Strong security with multi-layer safety systems. 30% fewer hallucinations improves output safety." @@ -263,7 +263,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -277,7 +277,7 @@ } ], "methodology": "Policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -291,7 +291,7 @@ } ], "methodology": "Terms of service review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -305,7 +305,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -319,7 +319,7 @@ } ], "methodology": "Verification of certifications", - "last_verified": "2026-01-14", + "last_verified": "2026-07-09", "notes": "Not HIPAA eligible" }, "zero_data_retention": { @@ -334,7 +334,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Good privacy with 30-day default retention. Zero retention for enterprise. Not HIPAA eligible." @@ -355,7 +355,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 92, @@ -369,7 +369,7 @@ } ], "methodology": "Factual accuracy testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -383,7 +383,7 @@ } ], "methodology": "Bias benchmarks and demographic testing", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 89, @@ -397,7 +397,7 @@ } ], "methodology": "Qualitative confidence expression", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 93, @@ -411,7 +411,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -425,7 +425,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "guardrails": { "score": 92, @@ -439,7 +439,7 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Excellent transparency with 30% fewer hallucinations. Thinking variant provides reasoning insight. Comprehensive system card." @@ -460,7 +460,7 @@ } ], "methodology": "API design and feature review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 95, @@ -474,7 +474,7 @@ } ], "methodology": "SDK quality and maintenance review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 91, @@ -490,11 +490,17 @@ "source": "OpenAI: Introducing GPT-5.5", "url": "https://openai.com/index/introducing-gpt-5-5/", "date": "2026-06-10", - "value": "GPT-5.2 superseded by GPT-5.4 (2026-03-05) and GPT-5.5 (2026-04-23, current flagship)" + "value": "GPT-5.2 superseded by GPT-5.4 (2026-03-05) and GPT-5.5 (2026-04-23)" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-07-09", + "value": "gpt-5.2-chat-latest API shutdown 2026-08-10 (announced 2026-05-08), replacement gpt-5.5; base gpt-5.2 snapshots remain served ('previous frontier model' per model page) with no announced shutdown" } ], "methodology": "Versioning policy review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -508,7 +514,7 @@ } ], "methodology": "Observability tools review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "support_quality": { "score": 94, @@ -522,7 +528,7 @@ } ], "methodology": "Support and documentation assessment", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 97, @@ -536,7 +542,7 @@ } ], "methodology": "Ecosystem breadth and depth analysis", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -550,7 +556,7 @@ } ], "methodology": "License terms review", - "last_verified": "2026-01-14" + "last_verified": "2026-07-09" } }, "notes": "Industry-leading operational maturity with largest ecosystem. Three model variants for different use cases. Excellent tooling." @@ -626,7 +632,7 @@ "1.4x price increase over GPT-5.1 ($1.75/$14)", "Slightly behind Claude Opus 4.5 on SWE-bench (80% vs 80.9%)", "Smaller context than Gemini 3 (400K vs 1M)", - "SUPERSEDED: GPT-5.4 (2026-03-05) and GPT-5.5 (2026-04-23) are newer; GPT-5.5 is OpenAI's current flagship" + "SUPERSEDED: GPT-5.4 (2026-03-05), GPT-5.5 (2026-04-23), and GPT-5.6 (2026-07-09) are newer; gpt-5.2-chat-latest API shutdown 2026-08-10" ], "best_for": [ @@ -648,8 +654,8 @@ "pricing": { "input": "$1.75 per 1M tokens", "output": "$14.00 per 1M tokens", - "notes": "1.4x price increase from GPT-5.1, reflecting enhanced capabilities", - "last_verified": "2026-01-14" + "notes": "Confirmed unchanged on official model page 2026-07-09 (snapshots gpt-5.2, gpt-5.2-2025-12-11).", + "last_verified": "2026-07-09" }, "context_window": 400000, "max_output": 128000, diff --git a/data/models/gpt-5-3-codex.json b/data/models/gpt-5-3-codex.json index af7c366..e72039c 100644 --- a/data/models/gpt-5-3-codex.json +++ b/data/models/gpt-5-3-codex.json @@ -4,9 +4,9 @@ "name": "GPT-5.3-Codex", "provider": "OpenAI", "version": "gpt-5-3-codex-2026-02-05", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's agentic coding specialist: ~80% SWE-bench Verified, 77.3% Terminal-Bench, SOTA on SWE-Bench Pro at release, ~25% faster than GPT-5.2-Codex. The 5.3 generation was Codex-only — there is no general-purpose GPT-5.3.", + "description": "OpenAI's agentic coding specialist, still active with no announced shutdown (as of 2026-07-09): ~80% SWE-bench Verified, 77.3% Terminal-Bench, SOTA on SWE-Bench Pro at release, ~25% faster than GPT-5.2-Codex. The 5.3 generation shipped no general-purpose API flagship — only this Codex model plus a gpt-5.3-chat-latest ChatGPT alias that shuts down 2026-08-10.", "website": "https://openai.com/index/introducing-gpt-5-3-codex/", "trust_vector": { @@ -37,7 +37,7 @@ } ], "methodology": "Industry-standard agentic coding benchmarks measuring real-world software engineering tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 88, @@ -51,7 +51,7 @@ } ], "methodology": "Reasoning benchmark review relative to general-purpose flagships", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 85, @@ -65,7 +65,7 @@ } ], "methodology": "Comparison against general-purpose models on non-coding workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 92, @@ -79,7 +79,7 @@ } ], "methodology": "Provider-reported reliability on multi-step agentic coding sessions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "~25% faster than GPT-5.2-Codex", @@ -93,21 +93,21 @@ } ], "methodology": "Provider-reported relative latency on agentic coding workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "400,000 tokens", - "confidence": "medium", + "confidence": "high", "evidence": [ { - "source": "OpenRouter Model Page", - "url": "https://openrouter.ai/openai/gpt-5.3-codex", - "date": "2026-02-05", - "value": "400K token context window listed for gpt-5.3-codex" + "source": "OpenAI Model Page: gpt-5.3-codex", + "url": "https://developers.openai.com/api/docs/models/gpt-5.3-codex", + "date": "2026-07-09", + "value": "400K token context window; 128K max output tokens (official specification)" } ], - "methodology": "Third-party model registry specification", - "last_verified": "2026-06-10" + "methodology": "Official specification from provider", + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -121,7 +121,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Best-in-class agentic coding at release (~80% SWE-bench Verified, 77.3% Terminal-Bench, SOTA SWE-Bench Pro). Specialized model — general-purpose accuracy intentionally trails flagships." @@ -142,7 +142,7 @@ } ], "methodology": "Testing against OWASP LLM01 attacks including coding-agent vectors", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 89, @@ -156,7 +156,7 @@ } ], "methodology": "Adversarial prompt testing against jailbreak datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 84, @@ -170,7 +170,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 90, @@ -184,7 +184,7 @@ } ], "methodology": "Safety testing across harmful content and dangerous-action categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 86, @@ -198,7 +198,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Good security posture with sandboxing in Codex environments. Autonomous code execution warrants strict permissioning and review gates in production pipelines." @@ -219,7 +219,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -233,7 +233,7 @@ } ], "methodology": "Policy review of data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days (zero retention available)", @@ -247,7 +247,7 @@ } ], "methodology": "Terms of service and enterprise documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -261,7 +261,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -275,7 +275,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 88, @@ -289,7 +289,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI enterprise posture. Proprietary source code sent as context is covered by no-training-by-default; zero-data-retention recommended for sensitive codebases." @@ -310,7 +310,7 @@ } ], "methodology": "Evaluation of reasoning and action transparency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -324,7 +324,7 @@ } ], "methodology": "Code correctness evaluation with execution-based verification", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 83, @@ -338,7 +338,7 @@ } ], "methodology": "Bias benchmarks and demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -352,7 +352,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in agentic outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 90, @@ -366,7 +366,7 @@ } ], "methodology": "Documentation completeness and clarity review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -380,7 +380,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -394,7 +394,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Agent transcripts give strong action-level auditability. Execution-based verification reduces unchecked hallucination relative to chat-style code generation." @@ -415,7 +415,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 95, @@ -429,7 +429,7 @@ } ], "methodology": "SDK quality, documentation, and maintenance review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -443,7 +443,7 @@ } ], "methodology": "Review of versioning policy and deprecation practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -457,7 +457,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 93, @@ -471,7 +471,7 @@ } ], "methodology": "Support and documentation assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 96, @@ -485,7 +485,7 @@ } ], "methodology": "Ecosystem breadth and depth analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -499,7 +499,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Deep coding-tool ecosystem (CLI, IDE, cloud agents). Codex line moves fast: GPT-5.2-Codex shuts down 2026-07-23, so plan for shorter model lifecycles than general flagships." @@ -540,9 +540,8 @@ "limitations": [ "Specialized for coding — weaker than flagships on general reasoning and writing", - "No general-purpose GPT-5.3 exists; the 5.3 generation was Codex-only", + "No general-purpose GPT-5.3 API flagship; only the Codex model and a gpt-5.3-chat-latest alias (shutdown 2026-08-10)", "Fast Codex lifecycle: predecessor GPT-5.2-Codex shuts down 2026-07-23, suggesting shorter support horizons", - "Pricing confirmed primarily via third-party listings (medium confidence)", "Not HIPAA eligible; 30-day default retention", "Autonomous code execution requires sandboxing and review gates" ], @@ -562,10 +561,10 @@ "metadata": { "pricing": { - "input": "$1.75 per 1M tokens (approximate)", - "output": "$14.00 per 1M tokens (approximate)", - "notes": "Pricing per OpenRouter listing (https://openrouter.ai/openai/gpt-5.3-codex); confidence medium pending first-party pricing page confirmation.", - "last_verified": "2026-06-10" + "input": "$1.75 per 1M tokens", + "output": "$14.00 per 1M tokens", + "notes": "Confirmed on OpenAI's official pricing and model pages 2026-07-09. Cached input $0.175 per 1M; Priority tier $3.50/$28.00 per 1M.", + "last_verified": "2026-07-09" }, "context_window": 400000, "max_output": 128000, @@ -588,7 +587,7 @@ "open_source": false, "architecture": "Transformer-based, fine-tuned for agentic software engineering (Codex line)", "parameters": "Not disclosed", - "knowledge_cutoff": "Late 2025" + "knowledge_cutoff": "August 31, 2025" }, "related_entities": ["gpt-5-2-codex", "gpt-5-5", "gpt-5-4", "claude-opus-4-8"], diff --git a/data/models/gpt-5-4.json b/data/models/gpt-5-4.json index f4b6537..f5c6d52 100644 --- a/data/models/gpt-5-4.json +++ b/data/models/gpt-5-4.json @@ -4,9 +4,9 @@ "name": "GPT-5.4", "provider": "OpenAI", "version": "gpt-5-4-2026-03-05", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's previous-generation flagship (superseded by GPT-5.5 in April 2026). Headline native computer use with 75% OSWorld-Verified, ~33% fewer factual errors than GPT-5.2, ~1.05M context. Thinking, Pro, mini, and nano variants.", + "description": "OpenAI's previous-generation flagship (superseded by GPT-5.5 in April 2026 and the GPT-5.6 family in July 2026). Fully supported with no announced shutdown; gpt-5.4-mini and gpt-5.4-nano are OpenAI's designated replacements for retiring GPT-5 mini/nano and gpt-4.1-nano. Headline native computer use with 75% OSWorld-Verified, ~33% fewer factual errors than GPT-5.2, ~1.05M context. Thinking, Pro, mini, and nano variants.", "website": "https://openai.com/index/introducing-gpt-5-4/", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Industry-standard coding benchmarks and provider-reported comparisons", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 95, @@ -39,7 +39,7 @@ } ], "methodology": "PhD-level and Olympiad-level reasoning benchmarks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 95, @@ -59,7 +59,7 @@ } ], "methodology": "Computer-use benchmarks and factual accuracy testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 94, @@ -73,7 +73,7 @@ } ], "methodology": "Internal testing across model variants", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.8s (standard) / faster on mini and nano", @@ -87,7 +87,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.5s (standard) / higher for Thinking and Pro", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "~1,050,000 tokens input / 128,000 output", @@ -115,7 +115,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -129,7 +129,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong generation defined by native computer use (75% OSWorld-Verified, +27.7 points over GPT-5.2) and a ~33% factual-error reduction. Superseded by GPT-5.5 six weeks after release." @@ -150,7 +150,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 90, @@ -164,7 +164,7 @@ } ], "methodology": "Adversarial prompt testing against jailbreak datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 84, @@ -178,7 +178,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 91, @@ -192,7 +192,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 86, @@ -206,7 +206,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Solid security posture with computer-use-specific guardrails. Native computer use expands the attack surface (UI-based injection) relative to text-only models." @@ -227,7 +227,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -241,7 +241,7 @@ } ], "methodology": "Policy review of data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days (zero retention available)", @@ -255,7 +255,7 @@ } ], "methodology": "Terms of service and enterprise documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -269,7 +269,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -283,7 +283,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 88, @@ -297,7 +297,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI enterprise posture: SOC 2, no API-data training by default, zero-data-retention options. Not HIPAA eligible." @@ -318,7 +318,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 93, @@ -332,7 +332,7 @@ } ], "methodology": "Factual accuracy testing on QA datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 85, @@ -346,7 +346,7 @@ } ], "methodology": "Bias benchmarks and demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 89, @@ -360,7 +360,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -374,7 +374,7 @@ } ], "methodology": "Documentation completeness and clarity review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -388,7 +388,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 91, @@ -402,7 +402,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Marked transparency improvement via the ~33% factual-error reduction. Computer-use action logging aids auditability of agentic runs." @@ -423,7 +423,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 95, @@ -437,7 +437,7 @@ } ], "methodology": "SDK quality, documentation, and maintenance review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 90, @@ -448,10 +448,16 @@ "url": "https://developers.openai.com/api/docs/deprecations", "date": "2026-04-24", "value": "Clear deprecation schedule; GPT-5.4 remains supported but OpenAI designates GPT-5.5 as the migration target for the GPT-5.x line" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-07-09", + "value": "Re-verified: GPT-5.4 not on the deprecations list; gpt-5.4-mini and gpt-5.4-nano are the named replacements for gpt-5-mini/gpt-5-nano (shutdown 2026-12-11) and gpt-4.1-nano (shutdown 2026-10-23)" } ], "methodology": "Review of versioning policy and historical deprecation practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -465,7 +471,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 94, @@ -479,7 +485,7 @@ } ], "methodology": "Support and documentation assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 97, @@ -493,7 +499,7 @@ } ], "methodology": "Ecosystem breadth and depth analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -507,7 +513,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Excellent operational maturity with the full variant family. Procurement caveat: superseded by GPT-5.5 only six weeks after release; plan migrations accordingly." @@ -602,8 +608,8 @@ "pricing": { "input": "$2.50 per 1M tokens", "output": "$15.00 per 1M tokens", - "notes": "Applies up to 272K input tokens; pricing doubles above that threshold. Thinking, Pro, mini, and nano variants priced separately.", - "last_verified": "2026-06-10" + "notes": "Confirmed on official pricing page 2026-07-09 (Batch 50% discount). Variants: gpt-5.4-mini $0.75/$4.50, gpt-5.4-nano $0.20/$1.25, gpt-5.4-pro $30/$180 per 1M. Base-rate applies up to 272K input tokens; pricing doubles above that threshold (per 2026-06-10 verification).", + "last_verified": "2026-07-09" }, "context_window": 1050000, "max_output": 128000, diff --git a/data/models/gpt-5-5.json b/data/models/gpt-5-5.json index 32f320d..147b6cb 100644 --- a/data/models/gpt-5-5.json +++ b/data/models/gpt-5-5.json @@ -3,10 +3,10 @@ "type": "model", "name": "GPT-5.5", "provider": "OpenAI", - "version": "gpt-5-5-2026-04-24", - "last_evaluated": "2026-06-10", + "version": "gpt-5-5-2026-04-23", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's current flagship (codename 'Spud') and first fully retrained base model since GPT-4.5. ~1.05M context, 85.0% ARC-AGI-2, 93.6% GPQA Diamond, 58.6% SWE-Bench Pro. Designated migration target for most of the GPT-5.x line.", + "description": "OpenAI's flagship from April 2026 (codename 'Spud'), first fully retrained base model since GPT-4.5. Succeeded at the top of the lineup by the GPT-5.6 family (Sol/Terra/Luna, publicly released 2026-07-09) but fully supported with no announced shutdown, and still the designated migration target for most of the GPT-5.x line. ~1.05M context, 85.0% ARC-AGI-2, 93.6% GPQA Diamond, 58.6% SWE-Bench Pro.", "website": "https://openai.com/index/introducing-gpt-5-5/", "trust_vector": { @@ -31,7 +31,7 @@ } ], "methodology": "Industry-standard coding and terminal benchmarks measuring real-world software engineering tasks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 98, @@ -57,7 +57,7 @@ } ], "methodology": "PhD-level science, frontier mathematics, and abstract reasoning benchmarks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 97, @@ -77,7 +77,7 @@ } ], "methodology": "Expert-comparison knowledge work and computer-use benchmarks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 95, @@ -91,7 +91,7 @@ } ], "methodology": "Internal consistency testing reported by provider across reasoning effort levels", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.0s (standard) / longer with extended reasoning", @@ -105,7 +105,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "4.0s (standard reasoning effort)", @@ -119,7 +119,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "~1,050,000 tokens input / 128,000 output", @@ -133,7 +133,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -147,7 +147,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "First fully retrained base since GPT-4.5. State-of-the-art across reasoning (85.0% ARC-AGI-2, 93.6% GPQA) and agentic coding (82.7% Terminal-Bench 2.0). ~40% more token-efficient than GPT-5.4." @@ -168,7 +168,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 91, @@ -182,7 +182,7 @@ } ], "methodology": "Adversarial prompt testing against jailbreak datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -196,7 +196,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 92, @@ -210,7 +210,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 87, @@ -224,7 +224,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Mature multi-layer safety stack. Retrained base required full safety recalibration, which OpenAI reports as complete; long-tail agentic behaviors still being characterized by third parties." @@ -245,7 +245,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -259,7 +259,7 @@ } ], "methodology": "Policy review of data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days (zero retention available)", @@ -273,7 +273,7 @@ } ], "methodology": "Terms of service and enterprise documentation review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -287,7 +287,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -301,7 +301,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 88, @@ -315,7 +315,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI enterprise posture: SOC 2, no API-data training by default, 30-day default retention with zero-data-retention options." @@ -336,7 +336,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 93, @@ -350,7 +350,7 @@ } ], "methodology": "Factual accuracy testing on QA datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 86, @@ -364,7 +364,7 @@ } ], "methodology": "Bias benchmarks and demographic testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 90, @@ -378,7 +378,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 94, @@ -392,7 +392,7 @@ } ], "methodology": "Documentation completeness and clarity review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -406,7 +406,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 92, @@ -420,7 +420,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency with reasoning summaries and detailed release documentation. Training data disclosure remains at industry-standard (limited) level." @@ -441,7 +441,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 95, @@ -455,7 +455,7 @@ } ], "methodology": "SDK quality, documentation, and maintenance review", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 92, @@ -466,10 +466,22 @@ "url": "https://developers.openai.com/api/docs/deprecations", "date": "2026-04-24", "value": "Published deprecation schedule; GPT-5.5 is the designated migration target for most of the GPT-5.x line" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-07-09", + "value": "Re-verified: GPT-5.5 remains the named replacement for gpt-5, gpt-5.1, gpt-5.2-chat-latest, gpt-5.2-codex, and gpt-5.3-chat-latest deprecations; no GPT-5.5 deprecation announced" + }, + { + "source": "OpenAI: Previewing GPT-5.6 Sol", + "url": "https://openai.com/index/previewing-gpt-5-6-sol/", + "date": "2026-07-09", + "value": "GPT-5.6 family (Sol/Terra/Luna) previewed 2026-06-26 under U.S. government-requested partner-only restrictions and publicly released 2026-07-09; Sol priced $5/$30, Terra $2.50/$15, Luna $1/$6 per 1M tokens" } ], "methodology": "Review of versioning policy and historical deprecation practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -483,7 +495,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 94, @@ -497,7 +509,7 @@ } ], "methodology": "Support and documentation assessment", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 97, @@ -511,7 +523,7 @@ } ], "methodology": "Ecosystem breadth and depth analysis", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -525,10 +537,10 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Industry-leading operational maturity. As the designated GPT-5.x migration target, GPT-5.5 offers the longest expected support horizon in the OpenAI lineup." + "notes": "Industry-leading operational maturity. As the designated GPT-5.x migration target, GPT-5.5 offers a long expected support horizon even after the GPT-5.6 family (Sol/Terra/Luna) launched 2026-07-09." } }, @@ -597,6 +609,7 @@ "limitations": [ "Premium pricing: $5/$30 per 1M tokens (2x GPT-5.4's base rate)", + "No longer the newest model: GPT-5.6 family released 2026-07-09; GPT-5.6 Terra is reported competitive with GPT-5.5 at half the price ($2.50/$15)", "Not HIPAA eligible", "30-day default API data retention (zero retention requires enterprise arrangement)", "GPT-5.5 Pro is very expensive ($30/$180 per 1M)", @@ -622,8 +635,8 @@ "pricing": { "input": "$5.00 per 1M tokens", "output": "$30.00 per 1M tokens", - "notes": "Batch and Flex processing at 50% discount. GPT-5.5 Pro priced at $30/$180 per 1M tokens.", - "last_verified": "2026-06-10" + "notes": "Confirmed on official pricing page 2026-07-09. Cached input $0.50 per 1M; Batch at 50% discount ($2.50/$15). GPT-5.5 Pro priced at $30/$180 per 1M tokens.", + "last_verified": "2026-07-09" }, "context_window": 1050000, "max_output": 128000, diff --git a/data/models/gpt-5.json b/data/models/gpt-5.json index 87f6304..84db547 100644 --- a/data/models/gpt-5.json +++ b/data/models/gpt-5.json @@ -3,10 +3,10 @@ "type": "model", "name": "GPT-5", "provider": "OpenAI", - "version": "gpt-5-1210", - "last_evaluated": "2025-11-07", + "version": "gpt-5-2025-08-07", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "OpenAI's latest flagship model with unified thinking capabilities, multimodal understanding, and enhanced reasoning. Successor to GPT-4o series.", + "description": "DEPRECATED: retired from ChatGPT (GPT-5 Instant/Thinking removed by 2026-02-13); gpt-5-chat-latest and gpt-5-codex API shut down 2026-07-23; the gpt-5-2025-08-07 snapshot's API shuts down 2026-12-11 (announced 2026-06-11) — migrate to GPT-5.5. Historically OpenAI's flagship model with unified thinking capabilities, multimodal understanding, and enhanced reasoning; successor to the GPT-4o series.", "website": "https://openai.com/gpt-5", "trust_vector": { @@ -25,7 +25,7 @@ } ], "methodology": "Standard coding benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -45,7 +45,7 @@ } ], "methodology": "Graduate and PhD-level reasoning benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 96, @@ -59,7 +59,7 @@ } ], "methodology": "Crowdsourced blind comparisons", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 94, @@ -73,7 +73,7 @@ } ], "methodology": "Internal testing across temperature settings", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.2s", @@ -87,7 +87,7 @@ } ], "methodology": "Platform-wide performance metrics", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.8s", @@ -101,21 +101,21 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "context_window": { - "value": "128,000 tokens", + "value": "400,000 tokens", "confidence": "high", "evidence": [ { - "source": "OpenAI Documentation", - "url": "https://platform.openai.com/docs/models/gpt-5", - "date": "2025-01-01", - "value": "128K token context window" + "source": "OpenAI Model Page: gpt-5", + "url": "https://developers.openai.com/api/docs/models/gpt-5", + "date": "2026-07-09", + "value": "400K token context window; 128K max output tokens" } ], "methodology": "Official specification", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -129,7 +129,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Top-tier performance across all dimensions. Unified thinking system enables more consistent and reliable outputs. Lower latency than competitors." @@ -150,7 +150,7 @@ } ], "methodology": "Testing against OWASP LLM01 attacks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -164,7 +164,7 @@ } ], "methodology": "Adversarial prompt testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 82, @@ -178,7 +178,7 @@ } ], "methodology": "Policy review and data handling practices", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_safety": { "score": 90, @@ -192,7 +192,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -206,7 +206,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Strong security with improved jailbreak resistance. Multi-layered safety systems provide robust output filtering." @@ -227,7 +227,7 @@ } ], "methodology": "Review of enterprise documentation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -241,7 +241,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -255,7 +255,7 @@ } ], "methodology": "Terms of service review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Zero retention available for enterprise customers" }, "pii_handling": { @@ -270,7 +270,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -284,7 +284,7 @@ } ], "methodology": "Verification of certifications", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Not HIPAA eligible currently" }, "zero_data_retention": { @@ -299,7 +299,7 @@ } ], "methodology": "Enterprise feature review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Good privacy posture with strong enterprise controls. 30-day default retention (vs Anthropic's 0-day). Not HIPAA eligible." @@ -320,7 +320,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -334,7 +334,7 @@ } ], "methodology": "Factual accuracy testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -348,7 +348,7 @@ } ], "methodology": "Bias benchmarks and demographic testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 87, @@ -362,7 +362,7 @@ } ], "methodology": "Qualitative confidence expression", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -376,7 +376,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -390,7 +390,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "guardrails": { "score": 91, @@ -404,7 +404,7 @@ } ], "methodology": "Safety mechanism analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Excellent transparency with unified thinking feature and comprehensive system card. Industry-leading hallucination prevention." @@ -425,7 +425,7 @@ } ], "methodology": "API design and feature review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -439,10 +439,10 @@ } ], "methodology": "SDK quality and maintenance review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "versioning_policy": { - "score": 90, + "score": 80, "confidence": "high", "evidence": [ { @@ -450,10 +450,16 @@ "url": "https://platform.openai.com/docs/api-reference/models", "date": "2025-01-01", "value": "Clear versioning with deprecation notices" + }, + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-07-09", + "value": "gpt-5-chat-latest and gpt-5-codex shutdown 2026-07-23 (announced 2026-04-22); gpt-5-2025-08-07, gpt-5-mini-2025-08-07, gpt-5-nano-2025-08-07 shutdown 2026-12-11 (announced 2026-06-11); recommended replacements gpt-5.5 / gpt-5.4-mini / gpt-5.4-nano" } ], "methodology": "Versioning policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 92, @@ -467,7 +473,7 @@ } ], "methodology": "Observability tools review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "support_quality": { "score": 93, @@ -481,7 +487,7 @@ } ], "methodology": "Support and documentation assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 96, @@ -495,7 +501,7 @@ } ], "methodology": "Ecosystem breadth and depth analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -509,10 +515,10 @@ } ], "methodology": "License terms review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, - "notes": "Industry-leading operational maturity with the most mature ecosystem. Excellent APIs, SDKs, and tooling." + "notes": "Industry-leading operational maturity with the most mature ecosystem. Versioning score reduced to reflect deprecation: gpt-5-chat-latest/gpt-5-codex shut down 2026-07-23 and the gpt-5-2025-08-07 snapshot shuts down 2026-12-11; migrate to GPT-5.5." } }, @@ -570,9 +576,9 @@ }, "strengths": [ - "Highest overall performance (LMSYS #1, 1342 ELO)", + "Top-tier performance at release (LMSYS #1, 1342 ELO in early 2025)", "Unified thinking system for enhanced reasoning", - "Lowest latency among frontier models (~1.2s p50)", + "Low latency (~1.2s p50)", "Most mature ecosystem (Assistants API, GPTs, plugins)", "Excellent multimodal capabilities (text, vision, audio)", "Superior observability and monitoring tools" @@ -581,9 +587,9 @@ "limitations": [ "Not HIPAA eligible (unlike Claude models)", "30-day data retention vs Anthropic's 0-day default", - "Smaller context window (128K vs Claude's 200K)", "Premium pricing comparable to Claude", - "Slightly behind Claude on specialized coding benchmarks" + "Slightly behind Claude on specialized coding benchmarks", + "DEPRECATED: retired from ChatGPT; gpt-5-chat-latest/gpt-5-codex API shutdown 2026-07-23; gpt-5-2025-08-07 snapshot API shutdown 2026-12-11 — migrate to GPT-5.5" ], "best_for": [ @@ -595,20 +601,22 @@ ], "not_recommended_for": [ + "New projects (deprecated; use GPT-5.5 instead)", "HIPAA-compliant healthcare applications", "Applications requiring zero data retention", - "Ultra-long document analysis (>128K tokens)", + "Ultra-long document analysis (>400K tokens)", "Specialized complex software engineering (prefer Claude Sonnet)" ], "metadata": { "pricing": { - "input": "$2.50 per 1M tokens", - "output": "$20.00 per 1M tokens", - "notes": "Priority tier pricing, batch API offers 50% discount", - "last_verified": "2025-11-09" + "input": "$1.25 per 1M tokens", + "output": "$10.00 per 1M tokens", + "notes": "Cached input $0.125 per 1M. Confirmed on official model page 2026-07-09; model deprecated (snapshot shutdown 2026-12-11).", + "last_verified": "2026-07-09" }, - "context_window": 128000, + "context_window": 400000, + "max_output": 128000, "languages": [ "English", "Spanish", @@ -631,9 +639,10 @@ "parameters": "Not disclosed" }, - "related_entities": ["claude-sonnet-4-5", "claude-opus-4-1", "gemini-2-5-pro"], + "related_entities": ["gpt-5-5", "gpt-5-1", "claude-sonnet-4-5", "claude-opus-4-1", "gemini-2-5-pro"], "tags": [ + "deprecated", "general-purpose", "multimodal", "low-latency", diff --git a/data/models/gpt-oss-120b.json b/data/models/gpt-oss-120b.json index 23ec334..70368f9 100644 --- a/data/models/gpt-oss-120b.json +++ b/data/models/gpt-oss-120b.json @@ -4,7 +4,7 @@ "name": "GPT-OSS-120B", "provider": "OpenAI", "version": "20250805", - "last_evaluated": "2025-11-17", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "OpenAI's first open-weight model released August 2025. 117B total params (5.1B active), Apache 2.0 license. Matches o4-mini on many benchmarks. Runs in 80GB memory.", "website": "https://openai.com/index/introducing-gpt-oss/", @@ -31,7 +31,7 @@ } ], "methodology": "Competition coding and tool use benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 93, @@ -51,7 +51,7 @@ } ], "methodology": "Math competition benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 89, @@ -71,7 +71,7 @@ } ], "methodology": "General knowledge and domain-specific testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 87, @@ -85,7 +85,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.0s", @@ -99,7 +99,7 @@ } ], "methodology": "Median latency estimation", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.2s", @@ -113,10 +113,10 @@ } ], "methodology": "95th percentile from community benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "context_window": { - "value": "128,000 tokens", + "value": "131,072 tokens", "confidence": "high", "evidence": [ { @@ -124,10 +124,16 @@ "url": "https://openai.com/index/gpt-oss-model-card/", "date": "2025-08-05", "value": "128K context window natively supported" + }, + { + "source": "OpenAI Model Docs", + "url": "https://developers.openai.com/api/docs/models/gpt-oss-120b", + "date": "2026-07-09", + "value": "131,072-token context window and up to 131,072 max output tokens" } ], "methodology": "Official specification", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -141,7 +147,7 @@ } ], "methodology": "Self-hosting provides full control", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Flagship open-source performance. MoE architecture activates 5.1B of 117B params per token. Matches or beats o4-mini on most benchmarks." @@ -162,7 +168,7 @@ } ], "methodology": "OWASP LLM01 testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 81, @@ -173,10 +179,16 @@ "url": "https://huggingface.co/openai/gpt-oss-120b", "date": "2025-08-10", "value": "Standard resistance, self-host allows custom guardrails" + }, + { + "source": "Prefill Attack Study (arXiv 2602.14689)", + "url": "https://arxiv.org/abs/2602.14689", + "date": "2026-02-16", + "value": "Large empirical study finds prefill attacks consistently effective against all major contemporary open-weight models; large reasoning models show partial resistance but remain vulnerable to tailored strategies" } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 95, @@ -190,7 +202,7 @@ } ], "methodology": "Self-hosting analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_safety": { "score": 83, @@ -204,7 +216,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "api_security": { "score": 90, @@ -218,10 +230,10 @@ } ], "methodology": "Deployment security review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, - "notes": "Good base security. Self-hosting provides complete control over safety guardrails and data handling." + "notes": "Good base security. Self-hosting provides complete control over safety guardrails and data handling. Note: Feb 2026 research (arXiv 2602.14689) shows open-weight models broadly remain vulnerable to prefill attacks — pair self-hosted deployments with external guardrails. OpenAI's gpt-oss-safeguard (Oct 2025), an Apache-2.0 safety-classifier fine-tune of gpt-oss, can serve as a policy-based moderation layer." }, "privacy_compliance": { @@ -239,7 +251,7 @@ } ], "methodology": "Self-hosting analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 100, @@ -253,7 +265,7 @@ } ], "methodology": "Privacy model analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (self-controlled)", @@ -267,7 +279,7 @@ } ], "methodology": "Self-hosting review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 100, @@ -281,7 +293,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 95, @@ -295,7 +307,7 @@ } ], "methodology": "Compliance model review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 100, @@ -309,7 +321,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Perfect privacy when self-hosted. No data sent to OpenAI. Full compliance control. Ideal for regulated industries." @@ -330,7 +342,7 @@ } ], "methodology": "Reasoning transparency", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 85, @@ -344,7 +356,7 @@ } ], "methodology": "QA testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 83, @@ -358,7 +370,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -372,7 +384,7 @@ } ], "methodology": "Confidence assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 98, @@ -386,7 +398,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 90, @@ -400,7 +412,7 @@ } ], "methodology": "Training data disclosure review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "guardrails": { "score": 85, @@ -414,7 +426,7 @@ } ], "methodology": "Safety mechanism review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Exceptional transparency. Full chain-of-thought access. Complete model weights and architecture disclosed. Open-source enables auditing." @@ -435,7 +447,7 @@ } ], "methodology": "API compatibility review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 96, @@ -449,7 +461,7 @@ } ], "methodology": "SDK ecosystem review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 98, @@ -460,10 +472,16 @@ "url": "https://huggingface.co/openai/gpt-oss-120b", "date": "2025-08-05", "value": "Weights frozen, no deprecation risk" + }, + { + "source": "Hugging Face Repository", + "url": "https://huggingface.co/openai/gpt-oss-120b", + "date": "2026-07-09", + "value": "Repo last updated 2025-08-26; weights stable with no new revisions, remains freely downloadable under Apache 2.0" } ], "methodology": "Version stability analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -477,7 +495,7 @@ } ], "methodology": "Monitoring capability review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -491,7 +509,7 @@ } ], "methodology": "Support ecosystem assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 97, @@ -505,7 +523,7 @@ } ], "methodology": "Ecosystem breadth analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "license_terms": { "score": 100, @@ -519,7 +537,7 @@ } ], "methodology": "License review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Exceptional operational flexibility. Apache 2.0 enables commercial use. Massive deployment ecosystem. Self-host or use managed platforms." @@ -545,7 +563,7 @@ "data-analysis": { "overall": 89, "notes": "Excellent for data analysis. Keep sensitive data on-premises. Full chain-of-thought for transparency.", - "alternatives": ["claude-opus-4", "openai-o3"] + "alternatives": ["claude-opus-4", "gpt-5-5"] }, "research-assistant": { "overall": 90, @@ -565,7 +583,7 @@ "financial-analysis": { "overall": 91, "notes": "Excellent for finance. Outperforms o3-mini on math. Self-host proprietary financial data.", - "alternatives": ["claude-opus-4", "openai-o3"] + "alternatives": ["claude-opus-4", "gpt-5-5"] }, "education": { "overall": 88, @@ -616,9 +634,9 @@ "input": "Free (self-hosted)", "output": "Free (self-hosted)", "notes": "Infrastructure costs only: ~$2-4/hr for H100. Managed platforms vary. Free for download and commercial use under Apache 2.0.", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, - "context_window": 128000, + "context_window": 131072, "languages": [ "English", "Spanish", diff --git a/data/models/gpt-oss-20b.json b/data/models/gpt-oss-20b.json index f3de03a..90b0688 100644 --- a/data/models/gpt-oss-20b.json +++ b/data/models/gpt-oss-20b.json @@ -4,7 +4,7 @@ "name": "GPT-OSS-20B", "provider": "OpenAI", "version": "20250805", - "last_evaluated": "2025-11-17", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "OpenAI's edge-optimized open-weight model released August 2025. 21B total params (3.6B active), Apache 2.0 license. Matches o3-mini despite small size. Runs in 16GB memory (edge devices).", "website": "https://openai.com/index/introducing-gpt-oss/", @@ -31,7 +31,7 @@ } ], "methodology": "Competition coding and tool use benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 93, @@ -51,7 +51,7 @@ } ], "methodology": "Math competition benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 89, @@ -71,7 +71,7 @@ } ], "methodology": "General knowledge and domain-specific testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 87, @@ -85,7 +85,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.0s", @@ -99,7 +99,7 @@ } ], "methodology": "Median latency estimation", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.2s", @@ -113,10 +113,10 @@ } ], "methodology": "95th percentile from community benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "context_window": { - "value": "128,000 tokens", + "value": "131,072 tokens", "confidence": "high", "evidence": [ { @@ -124,10 +124,16 @@ "url": "https://openai.com/index/gpt-oss-model-card/", "date": "2025-08-05", "value": "128K context window natively supported" + }, + { + "source": "OpenAI Model Docs", + "url": "https://developers.openai.com/api/docs/models/gpt-oss-20b", + "date": "2026-07-09", + "value": "131,072-token context window and up to 131,072 max output tokens" } ], "methodology": "Official specification", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -141,10 +147,10 @@ } ], "methodology": "Self-hosting provides full control", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, - "notes": "Flagship open-source performance. MoE architecture activates 5.1B of 117B params per token. Matches or beats o4-mini on most benchmarks." + "notes": "Strong open-source performance for its size. MoE architecture activates 3.6B of 21B params per token. Matches or beats o3-mini on common benchmarks despite edge-class footprint." }, "security": { @@ -162,7 +168,7 @@ } ], "methodology": "OWASP LLM01 testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 81, @@ -170,13 +176,19 @@ "evidence": [ { "source": "Community Testing", - "url": "https://huggingface.co/openai/gpt-oss-120b", + "url": "https://huggingface.co/openai/gpt-oss-20b", "date": "2025-08-10", "value": "Standard resistance, self-host allows custom guardrails" + }, + { + "source": "Prefill Attack Study (arXiv 2602.14689)", + "url": "https://arxiv.org/abs/2602.14689", + "date": "2026-02-16", + "value": "Large empirical study finds prefill attacks consistently effective against all major contemporary open-weight models; large reasoning models show partial resistance but remain vulnerable to tailored strategies" } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 95, @@ -190,7 +202,7 @@ } ], "methodology": "Self-hosting analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "output_safety": { "score": 83, @@ -204,7 +216,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "api_security": { "score": 90, @@ -218,10 +230,10 @@ } ], "methodology": "Deployment security review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, - "notes": "Good base security. Self-hosting provides complete control over safety guardrails and data handling." + "notes": "Good base security. Self-hosting provides complete control over safety guardrails and data handling. Note: Feb 2026 research (arXiv 2602.14689) shows open-weight models broadly remain vulnerable to prefill attacks — pair self-hosted deployments with external guardrails. OpenAI's gpt-oss-safeguard (Oct 2025), an Apache-2.0 safety-classifier fine-tune of gpt-oss, can serve as a policy-based moderation layer." }, "privacy_compliance": { @@ -239,7 +251,7 @@ } ], "methodology": "Self-hosting analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 100, @@ -253,7 +265,7 @@ } ], "methodology": "Privacy model analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (self-controlled)", @@ -267,7 +279,7 @@ } ], "methodology": "Self-hosting review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 100, @@ -281,7 +293,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 95, @@ -295,7 +307,7 @@ } ], "methodology": "Compliance model review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 100, @@ -309,7 +321,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Perfect privacy when self-hosted. No data sent to OpenAI. Full compliance control. Ideal for regulated industries." @@ -330,7 +342,7 @@ } ], "methodology": "Reasoning transparency", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 85, @@ -344,7 +356,7 @@ } ], "methodology": "QA testing", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 83, @@ -358,7 +370,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -372,7 +384,7 @@ } ], "methodology": "Confidence assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 98, @@ -386,7 +398,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 90, @@ -400,7 +412,7 @@ } ], "methodology": "Training data disclosure review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "guardrails": { "score": 85, @@ -414,7 +426,7 @@ } ], "methodology": "Safety mechanism review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Exceptional transparency. Full chain-of-thought access. Complete model weights and architecture disclosed. Open-source enables auditing." @@ -435,7 +447,7 @@ } ], "methodology": "API compatibility review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 96, @@ -449,7 +461,7 @@ } ], "methodology": "SDK ecosystem review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 98, @@ -457,13 +469,19 @@ "evidence": [ { "source": "Open Weights", - "url": "https://huggingface.co/openai/gpt-oss-120b", + "url": "https://huggingface.co/openai/gpt-oss-20b", "date": "2025-08-05", "value": "Weights frozen, no deprecation risk" + }, + { + "source": "Hugging Face Repository", + "url": "https://huggingface.co/openai/gpt-oss-20b", + "date": "2026-07-09", + "value": "Weights stable with no new revisions since the August 2025 release; remains freely downloadable under Apache 2.0" } ], "methodology": "Version stability analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 94, @@ -477,7 +495,7 @@ } ], "methodology": "Monitoring capability review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -491,7 +509,7 @@ } ], "methodology": "Support ecosystem assessment", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 97, @@ -505,7 +523,7 @@ } ], "methodology": "Ecosystem breadth analysis", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, "license_terms": { "score": 100, @@ -519,7 +537,7 @@ } ], "methodology": "License review", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" } }, "notes": "Exceptional operational flexibility. Apache 2.0 enables commercial use. Massive deployment ecosystem. Self-host or use managed platforms." @@ -545,7 +563,7 @@ "data-analysis": { "overall": 89, "notes": "Excellent for data analysis. Keep sensitive data on-premises. Full chain-of-thought for transparency.", - "alternatives": ["claude-opus-4", "openai-o3"] + "alternatives": ["claude-opus-4", "gpt-5-5"] }, "research-assistant": { "overall": 90, @@ -565,7 +583,7 @@ "financial-analysis": { "overall": 91, "notes": "Excellent for finance. Outperforms o3-mini on math. Self-host proprietary financial data.", - "alternatives": ["claude-opus-4", "openai-o3"] + "alternatives": ["claude-opus-4", "gpt-5-5"] }, "education": { "overall": 88, @@ -615,9 +633,9 @@ "input": "Free (self-hosted)", "output": "Free (self-hosted)", "notes": "Infrastructure costs only: ~$0.50-1/hr for consumer GPUs. Can run on edge devices. Free for download and commercial use under Apache 2.0.", - "last_verified": "2025-11-17" + "last_verified": "2026-07-09" }, - "context_window": 128000, + "context_window": 131072, "languages": [ "English", "Spanish", diff --git a/data/models/grok-3-beta.json b/data/models/grok-3-beta.json index d13263d..fd046b0 100644 --- a/data/models/grok-3-beta.json +++ b/data/models/grok-3-beta.json @@ -4,9 +4,9 @@ "name": "Grok 3 [Beta]", "provider": "xAI", "version": "Beta", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "RETIRED: xAI retired Grok 3 on 2026-05-15; retired API slugs now silently redirect to Grok 4.3 at Grok 4.3 pricing. Historically xAI's flagship beta model with exceptional coding performance and real-time knowledge via X platform. Migrate to Grok 4.3 (current xAI flagship) or Grok 4.1.", + "description": "RETIRED: xAI retired Grok 3 on 2026-05-15; retired API slugs now silently redirect to Grok 4.3 at Grok 4.3 pricing. Historically xAI's flagship beta model with exceptional coding performance and real-time knowledge via X platform. Migrate to Grok 4.3 or the new flagship Grok 4.5 (released 2026-07-08). Note: xAI merged into SpaceX and rebranded as SpaceXAI in mid-2026.", "website": "https://x.ai/grok", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 94, @@ -50,7 +50,7 @@ } ], "methodology": "Advanced reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 95, @@ -70,7 +70,7 @@ } ], "methodology": "Crowdsourced comparisons and knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 90, @@ -84,7 +84,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Beta status may result in occasional inconsistencies" }, "latency_p50": { @@ -99,7 +99,7 @@ } ], "methodology": "Median latency for API requests", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.4s", @@ -113,7 +113,7 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -127,7 +127,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 96, @@ -141,7 +141,7 @@ } ], "methodology": "Historical uptime data", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Beta status, production SLA TBD" } }, @@ -162,7 +162,7 @@ } ], "methodology": "Testing against OWASP LLM01 attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 86, @@ -176,7 +176,7 @@ } ], "methodology": "Testing against adversarial prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 80, @@ -190,7 +190,7 @@ } ], "methodology": "Analysis of privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 84, @@ -204,7 +204,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Beta status, safety systems still evolving" }, "api_security": { @@ -219,7 +219,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good security posture for beta product. Strong resistance to attacks, but systems still maturing." @@ -239,7 +239,7 @@ } ], "methodology": "Review of documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 82, @@ -253,7 +253,7 @@ } ], "methodology": "Analysis of privacy policy", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -267,7 +267,7 @@ } ], "methodology": "Review of terms of service", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 76, @@ -281,7 +281,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 78, @@ -295,7 +295,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Beta status, certifications in progress" }, "zero_data_retention": { @@ -310,7 +310,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Evolving privacy practices for beta product. Compliance certifications in progress. 30-day data retention." @@ -330,7 +330,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 84, @@ -344,7 +344,7 @@ } ], "methodology": "Testing on factual QA datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 78, @@ -358,7 +358,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Beta status, bias mitigation evolving" }, "uncertainty_quantification": { @@ -373,7 +373,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 85, @@ -387,7 +387,7 @@ } ], "methodology": "Review of documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -401,7 +401,7 @@ } ], "methodology": "Review of public disclosures", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 84, @@ -415,7 +415,7 @@ } ], "methodology": "Analysis of safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency for beta product. Real-time X integration provides current information. Some aspects still evolving." @@ -435,7 +435,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 80, @@ -449,7 +449,7 @@ } ], "methodology": "Review of SDK quality", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "SDKs still maturing" }, "versioning_policy": { @@ -465,12 +465,12 @@ { "source": "xAI May 15 Retirement Migration Guide", "url": "https://docs.x.ai/developers/migration/may-15-retirement", - "date": "2026-06-10", - "value": "grok-3 retired 2026-05-15; retired slugs silently redirect to grok-4.3 at grok-4.3 pricing" + "date": "2026-07-09", + "value": "Re-confirmed: grok-3 among eight models retired 2026-05-15; retired slugs silently redirect to grok-4.3 and bill at grok-4.3 pricing ($1.25/$2.50 per 1M)" } ], "methodology": "Review of versioning", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 79, @@ -484,7 +484,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 80, @@ -498,7 +498,7 @@ } ], "methodology": "Assessment of support", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 66, @@ -512,7 +512,7 @@ } ], "methodology": "Analysis of ecosystem", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 88, @@ -526,7 +526,7 @@ } ], "methodology": "Review of licensing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Model retired 2026-05-15; retired slugs silently redirect to grok-4.3 at grok-4.3 pricing. Versioning and ecosystem scores reduced to reflect retirement." @@ -648,10 +648,10 @@ ], "metadata": { "pricing": { - "input": "Free for X Premium+ users", - "output": "Free for X Premium+ users", - "notes": "Free for X (Twitter) Premium+ subscribers, API pricing TBD", - "last_verified": "2025-11-09" + "input": "N/A (retired)", + "output": "N/A (retired)", + "notes": "Model retired 2026-05-15. Requests to the grok-3 slug are redirected to grok-4.3 and billed at grok-4.3 pricing ($1.25 input / $2.50 output per 1M tokens). Historically free for X Premium+ subscribers.", + "last_verified": "2026-07-09" }, "context_window": 128000, "languages": [ diff --git a/data/models/grok-4-1.json b/data/models/grok-4-1.json index 21c6081..9941235 100644 --- a/data/models/grok-4-1.json +++ b/data/models/grok-4-1.json @@ -4,9 +4,9 @@ "name": "Grok 4.1", "provider": "xAI", "version": "4.1 (2025-11-17)", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "xAI's late-2025 flagship that debuted #1 on LMArena Text (1483 Elo) and led EQ-Bench3 for emotional intelligence, with a 2M token context window. Now superseded by Grok 4.3; the grok-4-1-fast variants were retired on 2026-05-15.", + "description": "xAI's late-2025 flagship that debuted #1 on LMArena Text (1483 Elo) and led EQ-Bench3 for emotional intelligence, with a 2M token context window. Now two generations behind: superseded by Grok 4.3 (2026-04-30) and the new flagship Grok 4.5 (2026-07-08). The grok-4-1-fast variants were retired on 2026-05-15; xAI itself merged into SpaceX and rebranded as SpaceXAI in mid-2026.", "website": "https://x.ai/news", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Review of third-party benchmark aggregator data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 92, @@ -44,7 +44,7 @@ } ], "methodology": "Provider launch evaluations and independent benchmark leaderboards", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 94, @@ -64,7 +64,7 @@ } ], "methodology": "Crowdsourced arena comparisons and aggregator metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 89, @@ -78,7 +78,7 @@ } ], "methodology": "Review of provider claims and community repeated-prompt reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "~2.5s (standard); sub-second with 4.1 Fast", @@ -92,7 +92,7 @@ } ], "methodology": "Median latency from third-party API benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "~6.0s (standard)", @@ -106,7 +106,7 @@ } ], "methodology": "95th percentile response time from third-party benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "2,000,000 tokens", @@ -120,7 +120,7 @@ } ], "methodology": "Official specification reflected in aggregator listings", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 95, @@ -134,7 +134,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Released 2025-11-17 and #1 on LMArena Text at launch (1483 Elo) with EQ-Bench3 leadership. Superseded by Grok 4.3 as xAI's flagship; grok-4-1-fast variants retired 2026-05-15." @@ -154,7 +154,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 85, @@ -168,7 +168,7 @@ } ], "methodology": "Review of adversarial prompt testing and community reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 80, @@ -182,7 +182,7 @@ } ], "methodology": "Analysis of privacy policies and data handling commitments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 84, @@ -196,7 +196,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 84, @@ -210,7 +210,7 @@ } ], "methodology": "Review of API security features", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Solid baseline; xAI publishes less safety evaluation detail than Anthropic, OpenAI, or Google." @@ -230,7 +230,7 @@ } ], "methodology": "Review of provider documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 80, @@ -244,7 +244,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days (standard API)", @@ -258,7 +258,7 @@ } ], "methodology": "Review of terms and retention policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 76, @@ -272,7 +272,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 74, @@ -286,7 +286,7 @@ } ], "methodology": "Verification of compliance certifications", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 74, @@ -300,7 +300,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Same thinner-than-peers xAI compliance posture as the rest of the Grok line: SOC 2 but no HIPAA program." @@ -320,7 +320,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 86, @@ -334,7 +334,7 @@ } ], "methodology": "Review of provider factuality evaluations and community testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 76, @@ -348,7 +348,7 @@ } ], "methodology": "Review of bias disclosures and independent reporting", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 82, @@ -362,7 +362,7 @@ } ], "methodology": "Qualitative assessment of confidence expression", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 84, @@ -376,7 +376,7 @@ } ], "methodology": "Review of documentation completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -390,7 +390,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 82, @@ -404,7 +404,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Notable for launch emphasis on hallucination reduction and emotional intelligence (EQ-Bench3 leader)." @@ -424,7 +424,7 @@ } ], "methodology": "Review of API design and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 84, @@ -438,7 +438,7 @@ } ], "methodology": "Review of SDK quality and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 68, @@ -448,11 +448,11 @@ "source": "xAI Migration Guide (May 15 Retirement)", "url": "https://docs.x.ai/developers/migration/may-15-retirement", "date": "2026-05-15", - "value": "grok-4-1-fast variants retired 2026-05-15, about six months after launch, with retired slugs redirecting to grok-4.3" + "value": "Re-confirmed 2026-07-09: grok-4-1-fast-reasoning and grok-4-1-fast-non-reasoning are on the official 2026-05-15 retirement list (standard grok-4.1 is not); retired slugs redirect to grok-4.3" } ], "methodology": "Review of deprecation timeline; rapid retirement and silent redirection penalize lifecycle predictability", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Six-month lifespan for Fast variants is short for production planning" }, "monitoring_observability": { @@ -467,7 +467,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 80, @@ -481,7 +481,7 @@ } ], "methodology": "Assessment of documentation and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 82, @@ -495,7 +495,7 @@ } ], "methodology": "Analysis of third-party integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 86, @@ -509,10 +509,10 @@ } ], "methodology": "Review of licensing terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Solid operations during its run, but the 2026-05-15 retirement of Fast variants and supersession by Grok 4.3 make this a legacy choice for new builds." + "notes": "Solid operations during its run, but the 2026-05-15 retirement of Fast variants and supersession by Grok 4.3 — and now Grok 4.5 (2026-07-08) — make this a legacy choice for new builds. Provider merged into SpaceX and rebranded SpaceXAI in mid-2026." } }, "use_case_ratings": { @@ -575,7 +575,7 @@ "Fast variant offered very low-cost agentic inference ($0.20/$0.50 per 1M)" ], "limitations": [ - "Superseded by Grok 4.3 as xAI's flagship", + "Superseded by Grok 4.3 and now Grok 4.5 (2026-07-08) as xAI/SpaceXAI's flagships", "grok-4-1-fast variants retired 2026-05-15 (about six months after launch)", "Standard pricing (~$3/$15 per 1M) far above Grok 4.3's $1.25/$2.50", "Thin enterprise compliance posture; no HIPAA eligibility", @@ -597,8 +597,8 @@ "pricing": { "input": "$3.00 per 1M tokens", "output": "$15.00 per 1M tokens", - "notes": "Grok 4.1 Fast variant was $0.20/$0.50 per 1M tokens before its 2026-05-15 retirement. Standard 4.1 superseded by Grok 4.3 ($1.25/$2.50).", - "last_verified": "2026-06-10" + "notes": "Grok 4.1 Fast variant was $0.20/$0.50 per 1M tokens before its 2026-05-15 retirement. Standard 4.1 superseded by Grok 4.3 ($1.25/$2.50) and Grok 4.5 ($2.00/$6.00). Standard 4.1 pricing and continued availability not explicitly re-confirmed in July 2026 sources — verify before new procurement.", + "last_verified": "2026-07-09" }, "context_window": 2000000, "languages": [ @@ -619,7 +619,7 @@ "architecture": "Transformer-based with reasoning and agentic tool-calling (Fast variant)", "parameters": "Not disclosed", "release_date": "2025-11-17", - "lifecycle_status": "Superseded by Grok 4.3; grok-4-1-fast retired 2026-05-15" + "lifecycle_status": "Superseded by Grok 4.3 and Grok 4.5 (2026-07-08); grok-4-1-fast retired 2026-05-15" }, "related_entities": ["grok-4-3", "grok-3-beta", "gpt-5-4", "claude-sonnet-4-6", "gemini-3-1-pro"], "tags": [ diff --git a/data/models/grok-4-3.json b/data/models/grok-4-3.json index aafa6a8..05f9b45 100644 --- a/data/models/grok-4-3.json +++ b/data/models/grok-4-3.json @@ -4,9 +4,9 @@ "name": "Grok 4.3", "provider": "xAI", "version": "4.3", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "xAI's current flagship model released in early May 2026, with a 1M token context window, reasoning, function calling, and structured outputs at aggressive pricing ($1.25/$2.50 per 1M tokens). Strong frontier performance, but a thinner enterprise compliance posture than Anthropic, OpenAI, or Google.", + "description": "xAI's workhorse model (released 2026-04-30): 1M context, reasoning, function calling, and structured outputs at $1.25/$2.50 per 1M tokens. Superseded as flagship by Grok 4.5 (2026-07-08, $2/$6, 500K context) but remains served and is the redirect target for retired Grok slugs. Strong frontier performance, but thinner enterprise compliance than Anthropic/OpenAI/Google, and the provider (now SpaceXAI post-SpaceX merger) faces active regulatory investigations over Grok content safety.", "website": "https://docs.x.ai/developers/models/grok-4.3", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Review of provider documentation and third-party benchmark aggregators", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 94, @@ -44,7 +44,7 @@ } ], "methodology": "Review of reasoning benchmark results from provider and aggregators", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 93, @@ -64,7 +64,7 @@ } ], "methodology": "Crowdsourced arena comparisons and aggregator quality metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 90, @@ -78,7 +78,7 @@ } ], "methodology": "Review of structured output features and community reports of repeated-prompt behavior", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "~2.0s", @@ -92,7 +92,7 @@ } ], "methodology": "Median latency from third-party API benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "~5.0s", @@ -106,7 +106,7 @@ } ], "methodology": "95th percentile response time from third-party benchmarking; reasoning mode adds variance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens", @@ -120,7 +120,7 @@ } ], "methodology": "Official specification from provider documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 96, @@ -134,13 +134,13 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Frontier-class performance with a 1M context window and reasoning, function calling, and structured outputs. Release date sources conflict (2026-04-30 per OpenRouter vs 2026-05-06 per llm-stats); xAI documentation is treated as primary." + "notes": "Frontier-class performance with a 1M context window and reasoning, function calling, and structured outputs. Launch date now consistently reported as 2026-04-30 (some aggregators previously listed 2026-05-06). Grok 4.5 (2026-07-08) now leads xAI/SpaceXAI's lineup, but 4.3 remains served and price-advantaged." }, "security": { - "overall_score": 83, + "overall_score": 82, "criteria": { "prompt_injection_resistance": { "score": 84, @@ -154,7 +154,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection patterns and review of published safety material", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 85, @@ -168,7 +168,7 @@ } ], "methodology": "Review of adversarial prompt testing results and community jailbreak reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 80, @@ -182,10 +182,10 @@ } ], "methodology": "Analysis of privacy policies and data handling commitments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { - "score": 82, + "score": 78, "confidence": "medium", "evidence": [ { @@ -193,10 +193,17 @@ "url": "https://docs.x.ai/developers/models/grok-4.3", "date": "2026-05-06", "value": "Content moderation in place; xAI publishes less safety evaluation detail than Anthropic/OpenAI/Google" + }, + { + "source": "TechPolicy.Press — Regulators Are Going After Grok and X", + "url": "https://www.techpolicy.press/regulators-are-going-after-grok-and-x-just-not-together/", + "date": "2026-07-09", + "value": "Ofcom and the European Commission opened formal investigations, and Brazil issued a 30-day ultimatum, over Grok's mass generation of sexualized imagery including apparent minors (CCDH: 3M+ sexualized images in under two weeks)" } ], - "methodology": "Safety testing across harmful content categories and review of published evaluations", - "last_verified": "2026-06-10" + "methodology": "Safety testing across harmful content categories and review of published evaluations; score reduced 2026-07-09 to reflect the ongoing Grok content-safety crisis and regulatory findings against the provider's safety systems", + "last_verified": "2026-07-09", + "notes": "The incidents center on Grok's consumer image-generation products on X rather than the grok-4.3 text API itself, but they evidence weak provider-level output-safety governance; score lowered from 82 to 78" }, "api_security": { "score": 84, @@ -210,10 +217,10 @@ } ], "methodology": "Review of API security features and authentication mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Reasonable baseline security, but xAI publishes substantially less safety and red-team documentation than Anthropic, OpenAI, or Google." + "notes": "Reasonable baseline security, but xAI publishes substantially less safety and red-team documentation than Anthropic, OpenAI, or Google. Overall score reduced one point (83 to 82) on 2026-07-09 after the Grok content-safety crisis (Ofcom/European Commission investigations, Brazil ultimatum) exposed weak provider-level output-safety governance." }, "privacy_compliance": { "overall_score": 76, @@ -230,7 +237,7 @@ } ], "methodology": "Review of provider documentation and enterprise materials", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 80, @@ -244,7 +251,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days (standard API)", @@ -258,7 +265,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 76, @@ -272,7 +279,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 72, @@ -286,7 +293,7 @@ } ], "methodology": "Verification of compliance certifications against major enterprise provider baselines", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 74, @@ -300,13 +307,13 @@ } ], "methodology": "Review of data handling practices and enterprise contract options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "xAI's enterprise compliance posture remains thinner than Anthropic, OpenAI, or Google: SOC 2 in place but no HIPAA eligibility program and fewer regulated-industry attestations." }, "trust_transparency": { - "overall_score": 82, + "overall_score": 81, "criteria": { "explainability": { "score": 86, @@ -320,7 +327,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 85, @@ -334,7 +341,7 @@ } ], "methodology": "Review of provider claims and factual QA testing", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 76, @@ -348,7 +355,7 @@ } ], "methodology": "Review of bias benchmark disclosures and independent reporting", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 83, @@ -362,7 +369,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 86, @@ -376,7 +383,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -390,10 +397,10 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { - "score": 82, + "score": 78, "confidence": "medium", "evidence": [ { @@ -401,13 +408,20 @@ "url": "https://docs.x.ai/", "date": "2026-05-06", "value": "Built-in moderation with developer controls; lighter-touch defaults than peers" + }, + { + "source": "The Conversation — Grok sexualized images AI reckoning", + "url": "https://theconversation.com/the-furore-over-groks-sexualised-images-has-begun-an-ai-reckoning-275448", + "date": "2026-07-01", + "value": "2026 Grok controversy showed xAI's guardrails failed at scale on sexualized imagery, including of minors, prompting UK, EU, and Brazilian regulatory action" } ], - "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "methodology": "Analysis of built-in safety mechanisms; score reduced 2026-07-09 given demonstrated large-scale guardrail failures in the provider's deployed Grok products", + "last_verified": "2026-07-09", + "notes": "Score lowered from 82 to 78: guardrail failures occurred in Grok consumer/image products rather than this text API, but they materially weaken confidence in xAI's safety-engineering culture" } }, - "notes": "Good developer-facing documentation and inspectable reasoning, but less published safety/bias evaluation than major competitors." + "notes": "Good developer-facing documentation and inspectable reasoning, but less published safety/bias evaluation than major competitors. Overall score reduced one point (82 to 81) on 2026-07-09 to reflect the guardrails downgrade following the Grok content-safety crisis and resulting Ofcom/EC/Brazil regulatory actions." }, "operational_excellence": { "overall_score": 82, @@ -424,7 +438,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 84, @@ -438,7 +452,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 75, @@ -452,7 +466,7 @@ } ], "methodology": "Review of deprecation/migration practices; silent redirects of retired slugs reduce predictability for pinned workloads", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Silent redirection of retired model slugs to grok-4.3 can change behavior of production systems without explicit failure signals" }, "monitoring_observability": { @@ -467,7 +481,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 80, @@ -481,7 +495,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 82, @@ -495,7 +509,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 86, @@ -509,10 +523,10 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Strong API and pricing, but the May 2026 retirement wave (with silent slug redirects to grok-4.3) highlights an aggressive deprecation culture enterprises should plan around." + "notes": "Strong API and pricing, but the May 2026 retirement wave (with silent slug redirects to grok-4.3) highlights an aggressive deprecation culture enterprises should plan around. In mid-2026 xAI completed its merger into SpaceX and rebranded as SpaceXAI (new identity announced 2026-07-06); API endpoints and docs remain on x.ai domains as of 2026-07-09." } }, "use_case_ratings": { @@ -581,7 +595,8 @@ "Higher per-token rate applies above 200K context", "Limited published safety, bias, and red-team evaluation detail", "Zero-data-retention only via negotiated enterprise terms", - "Conflicting release-date records across aggregators reflect lighter release documentation" + "Provider turbulence: xAI merged into SpaceX and rebranded SpaceXAI in mid-2026, while under active regulatory investigation (Ofcom, European Commission; Brazil ultimatum) over Grok content-safety failures in 2026", + "No longer the flagship: Grok 4.5 (2026-07-08) sits above it in the lineup" ], "best_for": [ "Cost-sensitive agentic and coding workloads needing frontier quality", @@ -599,8 +614,8 @@ "pricing": { "input": "$1.25 per 1M tokens", "output": "$2.50 per 1M tokens", - "notes": "Cached input $0.20 per 1M tokens. Higher per-token rate applies for requests above 200K context.", - "last_verified": "2026-06-10" + "notes": "Cached input $0.20 per 1M tokens. Higher per-token rate applies for requests above 200K context. Tool calls billed separately (web/X search and code execution $5 per 1K calls). Re-confirmed against xAI docs July 2026.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "languages": [ @@ -621,11 +636,11 @@ "open_source": false, "architecture": "Transformer-based with native reasoning, function calling, and structured outputs", "parameters": "Not disclosed", - "release_date": "Early May 2026 (2026-04-30 per OpenRouter; 2026-05-06 per llm-stats)" + "release_date": "2026-04-30", + "lifecycle_status": "Served and supported; superseded as flagship by Grok 4.5 (2026-07-08). Retired legacy Grok slugs redirect here." }, "related_entities": ["grok-4-1", "grok-3-beta", "gpt-5-5", "claude-opus-4-8", "gemini-3-1-pro"], "tags": [ - "flagship", "reasoning", "long-context", "function-calling", diff --git a/data/models/kimi-k2-6.json b/data/models/kimi-k2-6.json index 78ca86e..4aa7118 100644 --- a/data/models/kimi-k2-6.json +++ b/data/models/kimi-k2-6.json @@ -4,9 +4,9 @@ "name": "Kimi K2.6", "provider": "Moonshot AI", "version": "20260420", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Moonshot AI's open-weight 1T-parameter MoE (32B active) with vendor-reported 80.2% SWE-Bench Verified and 58.6 SWE-Bench Pro. Agent Swarm orchestration scales to 300 sub-agents and 4,000 coordinated steps for long-horizon coding.", + "description": "Moonshot AI's open-weight 1T-parameter MoE (32B active) with vendor-reported 80.2% SWE-Bench Verified and 58.6 SWE-Bench Pro. Agent Swarm orchestration scales to 300 sub-agents and 4,000 coordinated steps for long-horizon coding. Remains Moonshot's general-purpose flagship as of July 2026; a coding-specialized sibling, Kimi K2.7-Code (built on K2.6, also open-weight Modified MIT), shipped 2026-06-12.", "website": "https://huggingface.co/moonshotai/Kimi-K2.6", "trust_vector": { @@ -37,7 +37,7 @@ } ], "methodology": "Vendor-reported industry-standard coding benchmarks; scores pending broad independent replication", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 90, @@ -51,7 +51,7 @@ } ], "methodology": "Vendor-reported tool-augmented reasoning benchmarks requiring multi-step problem solving", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 88, @@ -65,7 +65,7 @@ } ], "methodology": "Review of vendor benchmark suite and community evaluations across knowledge domains", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 85, @@ -79,7 +79,7 @@ } ], "methodology": "Community testing of repeated runs and long-horizon agent trajectories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "3.0s", @@ -93,7 +93,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes; self-hosted latency depends on hardware", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "7.5s", @@ -107,7 +107,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "262,144 tokens", @@ -121,7 +121,7 @@ } ], "methodology": "Official specification from model card", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 95, @@ -135,7 +135,7 @@ } ], "methodology": "Review of platform availability and self-hosting fallback options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Vendor-reported open-weight leadership on agentic coding (80.2% SWE-Bench Verified, 58.6 SWE-Bench Pro). Agent Swarm scales to 300 sub-agents / 4,000 coordinated steps. Most headline scores are vendor-reported and await independent replication." @@ -156,7 +156,7 @@ } ], "methodology": "Review of vendor safety documentation and community red-team reports against OWASP LLM01 patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 75, @@ -170,7 +170,7 @@ } ], "methodology": "Testing against adversarial prompt datasets; open-weight deployments inherit deployer responsibility", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 76, @@ -184,7 +184,7 @@ } ], "methodology": "Analysis of privacy policies and self-hosting data-control options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 80, @@ -198,7 +198,7 @@ } ], "methodology": "Safety testing across harmful content categories per vendor card and community reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 84, @@ -212,7 +212,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard open-model security posture. No published third-party security audit; self-hosting shifts security responsibility to the deployer." @@ -239,7 +239,7 @@ } ], "methodology": "Review of provider jurisdiction and third-party hosting options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 75, @@ -253,7 +253,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Per Moonshot policy on first-party API (China jurisdiction); zero when self-hosted", @@ -267,7 +267,7 @@ } ], "methodology": "Review of terms of service and deployment-dependent retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 72, @@ -281,7 +281,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 68, @@ -295,7 +295,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 85, @@ -309,7 +309,7 @@ } ], "methodology": "Review of self-hosting deployment options enabling zero retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "First-party API operates under Chinese jurisdiction — a material caveat for Western regulated industries. Open weights fully mitigate this for organizations able to self-host or use Western inference providers." @@ -330,7 +330,7 @@ } ], "methodology": "Evaluation of reasoning and agent-trajectory transparency", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -344,7 +344,7 @@ } ], "methodology": "Testing on factual QA datasets and tool-augmented workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 75, @@ -358,7 +358,7 @@ } ], "methodology": "Review of published bias benchmarks and community evaluations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 78, @@ -372,7 +372,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 88, @@ -386,7 +386,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 72, @@ -400,7 +400,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 78, @@ -414,7 +414,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Open weights and a detailed model card provide good architectural transparency; training data disclosure and independent benchmark verification remain limited." @@ -435,7 +435,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 82, @@ -449,7 +449,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 80, @@ -460,10 +460,16 @@ "url": "https://huggingface.co/moonshotai", "date": "2026-04-20", "value": "K2.6 supersedes K2.5/K2; prior weights remain available, but cadence is fast" + }, + { + "source": "MarkTechPost - Kimi K2.7-Code release", + "url": "https://www.marktechpost.com/2026/06/12/moonshot-ai-releases-kimi-k2-7-code-a-coding-model-reporting-21-8-on-kimi-code-bench-v2-over-k2-6/", + "date": "2026-06-12", + "value": "Kimi K2.7-Code released 2026-06-12: coding-specialized open-weight model built on K2.6 (1T/32B active, 256K context, Modified MIT) with ~30% lower reasoning-token usage; vendor claims +21.8% on Kimi Code Bench v2 over K2.6 — all K2.7 benchmarks are Moonshot-proprietary with no independent public-suite results yet" } ], "methodology": "Review of versioning practices and weight availability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 78, @@ -477,7 +483,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 78, @@ -491,7 +497,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 86, @@ -505,7 +511,7 @@ } ], "methodology": "Analysis of third-party hosting, integrations, and tooling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 84, @@ -519,7 +525,7 @@ } ], "methodology": "Review of licensing terms and restrictions; attribution clause is trust-relevant for large-scale commercial use", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong open-model ecosystem presence. Modified MIT license is permissive for most users but the attribution clause above 100M MAU / $20M monthly revenue requires legal review at hyperscale." @@ -529,7 +535,7 @@ "use_case_ratings": { "code-generation": { "overall": 95, - "notes": "Vendor-reported 80.2% SWE-Bench Verified and 58.6 SWE-Bench Pro; Agent Swarm excels at long-horizon multi-file engineering.", + "notes": "Vendor-reported 80.2% SWE-Bench Verified and 58.6 SWE-Bench Pro; Agent Swarm excels at long-horizon multi-file engineering. For pure coding workloads, Moonshot's coding-specialized K2.7-Code (June 2026, built on K2.6) claims further gains with ~30% lower token usage.", "alternatives": ["claude-opus-4-8", "glm-5", "gpt-5-5"] }, "customer-support": { @@ -613,10 +619,10 @@ "metadata": { "pricing": { - "input": "$0.95 per 1M tokens (approx.)", - "output": "$4.00 per 1M tokens (approx.)", - "notes": "First-party Moonshot API pricing; third-party hosts on OpenRouter vary. Self-hosting cost is infrastructure-dependent.", - "last_verified": "2026-06-10" + "input": "$0.95 per 1M tokens ($0.16 cache hit)", + "output": "$4.00 per 1M tokens", + "notes": "First-party Moonshot API pricing confirmed July 2026; cached input drops to $0.16 per 1M (~83% off). Third-party hosts on OpenRouter vary (some cheaper, e.g. $0.55/$2.00). Self-hosting cost is infrastructure-dependent.", + "last_verified": "2026-07-09" }, "context_window": 262144, "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], diff --git a/data/models/llama-3-1-405b.json b/data/models/llama-3-1-405b.json index 8f86c78..0d777f2 100644 --- a/data/models/llama-3-1-405b.json +++ b/data/models/llama-3-1-405b.json @@ -4,9 +4,9 @@ "name": "Llama 3.1 405B", "provider": "Meta", "version": "llama-3.1-405b", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Meta's largest open-source model with 405 billion parameters, offering complete transparency, self-hosting capabilities, and competitive performance with proprietary models. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026.", + "description": "Meta's largest open-source model with 405 billion parameters, offering complete transparency, self-hosting capabilities, and competitive performance with proprietary models. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models with Muse Spark (April 2026). Weights remain broadly available on Hugging Face and via many API hosts as of July 2026.", "website": "https://llama.meta.com", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 83, @@ -44,7 +44,7 @@ } ], "methodology": "Mathematical and scientific reasoning benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 87, @@ -58,7 +58,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 85, @@ -72,7 +72,7 @@ } ], "methodology": "Community evaluation and testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "Varies by deployment", @@ -86,7 +86,7 @@ } ], "methodology": "Third-party hosting performance", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Self-hosted performance varies significantly based on infrastructure" }, "latency_p95": { @@ -101,7 +101,7 @@ } ], "methodology": "Third-party hosting performance", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uptime_sla": { "value": "Deployment dependent", @@ -115,7 +115,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Self-hosting offers complete control over uptime" }, "context_window": { @@ -130,7 +130,7 @@ } ], "methodology": "Official model specifications", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "multimodal_support": { "value": "Text-only", @@ -144,7 +144,7 @@ } ], "methodology": "Official model capabilities", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } } }, @@ -163,7 +163,7 @@ } ], "methodology": "Safety testing and red teaming", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Open weights mean users can modify safety guardrails" }, "prompt_injection_defense": { @@ -178,7 +178,7 @@ } ], "methodology": "Community security testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 80, @@ -192,7 +192,7 @@ } ], "methodology": "Architecture review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Self-hosting provides complete control over data" }, "adversarial_robustness": { @@ -207,7 +207,7 @@ } ], "methodology": "Adversarial testing by Meta", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "content_filtering": { "score": 82, @@ -221,7 +221,7 @@ } ], "methodology": "Safety tooling review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Requires separate Llama Guard deployment" } } @@ -241,7 +241,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Zero external data transmission in self-hosted deployments" }, "gdpr_compliance": { @@ -256,7 +256,7 @@ } ], "methodology": "Privacy architecture review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "hipaa_eligible": { "score": 95, @@ -270,7 +270,7 @@ } ], "methodology": "Healthcare compliance assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "soc2_certified": { "score": 90, @@ -284,7 +284,7 @@ } ], "methodology": "Deployment architecture review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Users responsible for their own SOC 2 compliance" }, "data_sovereignty": { @@ -299,7 +299,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Best-in-class data sovereignty" }, "encryption_at_rest": { @@ -314,7 +314,7 @@ } ], "methodology": "Deployment architecture review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "encryption_in_transit": { "score": 95, @@ -328,7 +328,7 @@ } ], "methodology": "Deployment architecture review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } } }, @@ -347,7 +347,7 @@ } ], "methodology": "Documentation completeness review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 85, @@ -361,7 +361,7 @@ } ], "methodology": "Public documentation review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Good transparency on data size and composition" }, "safety_testing_transparency": { @@ -376,7 +376,7 @@ } ], "methodology": "Safety documentation review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "bias_evaluation": { "score": 90, @@ -390,7 +390,7 @@ } ], "methodology": "Bias benchmarks review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "decision_explainability": { "score": 100, @@ -404,7 +404,7 @@ } ], "methodology": "Model accessibility assessment", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Full model inspection possible" }, "versioning_changelog": { @@ -419,7 +419,7 @@ } ], "methodology": "Version management review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } } }, @@ -438,7 +438,7 @@ } ], "methodology": "Deployment options review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Maximum deployment flexibility" }, "api_reliability": { @@ -453,7 +453,7 @@ } ], "methodology": "Third-party API monitoring", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Varies by API provider" }, "rate_limits": { @@ -468,7 +468,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "cost_efficiency": { "score": 75, @@ -479,10 +479,16 @@ "url": "https://www.together.ai/pricing", "date": "2024-10-01", "value": "$3.00 per 1M input tokens via API, infrastructure costs for self-hosting" + }, + { + "source": "Price Per Token (multi-provider comparison)", + "url": "https://pricepertoken.com/pricing-page/model/meta-llama-llama-3.1-405b", + "date": "2026-07-09", + "value": "Still hosted by multiple providers; pricing ranges roughly $0.80-$9.50 per 1M tokens across hosts (DeepInfra cheapest)" } ], "methodology": "Cost analysis", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Free to use, but requires significant infrastructure for self-hosting (8x H100 GPUs minimum)" }, "monitoring_observability": { @@ -497,7 +503,7 @@ } ], "methodology": "Tooling availability assessment", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Requires custom monitoring implementation" }, "support_quality": { @@ -512,7 +518,7 @@ } ], "methodology": "Support channels review", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Relies on community support" } } @@ -534,7 +540,7 @@ "Safety guardrails can be modified (security consideration)", "Higher latency compared to smaller models", "Complex deployment and maintenance", - "Legacy status: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026, so future open updates are unlikely" + "Legacy status: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models (Muse Spark, April 2026), so future open updates are unlikely" ], "metadata": { "license": "Llama 3.1 Community License (open for commercial use)", diff --git a/data/models/llama-3-3-70b.json b/data/models/llama-3-3-70b.json index ec40580..115c226 100644 --- a/data/models/llama-3-3-70b.json +++ b/data/models/llama-3-3-70b.json @@ -4,9 +4,9 @@ "name": "Llama 3.3 70B", "provider": "Meta", "version": "2024-12", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Meta's powerful 70B parameter Llama 3.3 model offering strong performance with open-source flexibility and an excellent balance of capability and resource efficiency for self-hosted deployments. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026.", + "description": "Meta's powerful 70B parameter Llama 3.3 model offering strong performance with open-source flexibility and an excellent balance of capability and resource efficiency for self-hosted deployments. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models with Muse Spark (April 2026). Weights remain widely available and hosted as of July 2026.", "website": "https://llama.meta.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 82, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 76, @@ -52,7 +52,7 @@ } ], "methodology": "Knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 77, @@ -66,7 +66,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.4s", @@ -80,7 +80,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.8s", @@ -94,7 +94,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -108,7 +108,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 95, @@ -122,7 +122,7 @@ } ], "methodology": "Deployment-dependent", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong mathematical reasoning (77% MATH). Good balance for self-hosted deployments." @@ -142,7 +142,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 79, @@ -156,7 +156,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -170,7 +170,7 @@ } ], "methodology": "Deployment analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 80, @@ -184,7 +184,7 @@ } ], "methodology": "Safety benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 82, @@ -198,7 +198,7 @@ } ], "methodology": "Deployment review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good baseline security with self-hosted control." @@ -218,7 +218,7 @@ } ], "methodology": "Deployment analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 98, @@ -232,7 +232,7 @@ } ], "methodology": "Data flow analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "User-controlled", @@ -246,7 +246,7 @@ } ], "methodology": "Deployment analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 92, @@ -260,7 +260,7 @@ } ], "methodology": "Architecture review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -274,7 +274,7 @@ } ], "methodology": "Deployment options", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 98, @@ -288,7 +288,7 @@ } ], "methodology": "Deployment analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy with self-hosted deployment." @@ -308,7 +308,7 @@ } ], "methodology": "Reasoning evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 82, @@ -322,7 +322,7 @@ } ], "methodology": "Community evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 82, @@ -336,7 +336,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 83, @@ -350,7 +350,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 91, @@ -364,7 +364,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 87, @@ -378,7 +378,7 @@ } ], "methodology": "Technical documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 89, @@ -392,7 +392,7 @@ } ], "methodology": "Safety system review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency as open-source model." @@ -412,7 +412,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 87, @@ -426,7 +426,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -440,7 +440,7 @@ } ], "methodology": "Versioning review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 79, @@ -454,7 +454,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 83, @@ -468,7 +468,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 88, @@ -482,7 +482,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -496,7 +496,7 @@ } ], "methodology": "License review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good operational maturity with mature Llama ecosystem." @@ -507,7 +507,7 @@ "overall": 76, "notes": "Moderate coding capabilities. Better options for complex development.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "gpt-4-1" ] }, @@ -529,7 +529,7 @@ "overall": 84, "notes": "Strong mathematical reasoning (77% MATH) for analysis.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "openai-o3" ] }, @@ -537,14 +537,14 @@ "overall": 78, "notes": "Good for research with solid knowledge base.", "alternatives": [ - "llama-4-behemoth" + "llama-4-maverick" ] }, "legal-compliance": { "overall": 82, "notes": "Good for legal with data sovereignty via self-hosting.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "claude-sonnet-4-5" ] }, @@ -552,7 +552,7 @@ "overall": 86, "notes": "Good for healthcare with self-hosted HIPAA compliance.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "claude-sonnet-4-5" ] }, @@ -560,7 +560,7 @@ "overall": 83, "notes": "Strong math capabilities for financial modeling.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "openai-o3" ] }, @@ -568,7 +568,7 @@ "overall": 82, "notes": "Good for education with strong mathematical reasoning.", "alternatives": [ - "llama-4-behemoth" + "llama-4-maverick" ] }, "creative-writing": { @@ -594,7 +594,7 @@ "No managed API from Meta", "Deployment expertise needed", "Uptime depends on hosting", - "Legacy status: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models in 2026, so future open updates are unlikely" + "Legacy status: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models (Muse Spark, April 2026), so future open updates are unlikely" ], "best_for": [ "Self-hosted deployments requiring data sovereignty", @@ -615,7 +615,8 @@ "pricing": { "input": "Self-hosted (infrastructure costs)", "output": "Self-hosted (infrastructure costs)", - "notes": "Open-source. Typically $0.30-1.00 per 1M tokens with optimized deployment." + "notes": "Open-source. Typically $0.30-1.00 per 1M tokens with optimized deployment; hosted APIs (DeepInfra, Together, etc.) remain in that range as of July 2026.", + "last_verified": "2026-07-09" }, "context_window": 128000, "languages": [ @@ -639,7 +640,7 @@ "parameters": "70B" }, "related_entities": [ - "llama-4-behemoth", + "llama-4-maverick", "llama-4-scout", "llama-3-1-405b" ], diff --git a/data/models/llama-4-behemoth.json b/data/models/llama-4-behemoth.json index 54e554c..c88063c 100644 --- a/data/models/llama-4-behemoth.json +++ b/data/models/llama-4-behemoth.json @@ -4,9 +4,9 @@ "name": "Llama 4 Behemoth", "provider": "Meta", "version": "2025-02", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Meta's announced 2T-total/288B-active parameter Llama 4 teacher model that was NEVER RELEASED. It remains 'announced, not released' as of June 2026: Meta gave no update when asked in January 2026 and has effectively exited open-weight frontier releases, shipping the proprietary closed-weight 'Muse Spark' (April 2026) instead. Scores reflect unverifiable preview-era claims; the model is not available for any deployment.", + "description": "Meta's announced 2T-total/288B-active parameter Llama 4 teacher model that was NEVER RELEASED. It remains 'announced, not released' as of July 2026 — effectively shelved (never formally cancelled): Meta gave no update when asked in January 2026 and has effectively exited open-weight frontier releases, shipping the proprietary closed-weight 'Muse Spark' (April 8, 2026) instead. Scores reflect unverifiable preview-era claims; the model is not available for any deployment.", "website": "https://llama.meta.com/llama-4", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 95, @@ -50,7 +50,7 @@ } ], "methodology": "Advanced mathematical and scientific reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 90, @@ -70,7 +70,7 @@ } ], "methodology": "Crowdsourced comparisons and knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 89, @@ -84,7 +84,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "2.8s", @@ -98,7 +98,7 @@ } ], "methodology": "Median latency on recommended hardware", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "5.2s", @@ -112,7 +112,7 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -126,7 +126,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 60, @@ -146,7 +146,7 @@ } ], "methodology": "User-controlled deployment", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Model never released; no deployment is possible" } }, @@ -167,7 +167,7 @@ } ], "methodology": "Testing against prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 81, @@ -181,7 +181,7 @@ } ], "methodology": "Testing against adversarial prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -195,7 +195,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 82, @@ -209,7 +209,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 84, @@ -223,7 +223,7 @@ } ], "methodology": "Review of deployment best practices", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Security controlled by deployment team" } }, @@ -244,7 +244,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 98, @@ -258,7 +258,7 @@ } ], "methodology": "Analysis of data flow", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "User-controlled", @@ -272,7 +272,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 92, @@ -286,7 +286,7 @@ } ], "methodology": "Review of deployment architecture", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -300,7 +300,7 @@ } ], "methodology": "Review of deployment options", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Can achieve any compliance requirement with proper deployment" }, "zero_data_retention": { @@ -315,7 +315,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy with self-hosted deployment. Full control over data residency, retention, and compliance. No data shared with Meta." @@ -335,7 +335,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 84, @@ -349,7 +349,7 @@ } ], "methodology": "Community evaluation and testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 83, @@ -363,7 +363,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 85, @@ -377,7 +377,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -391,7 +391,7 @@ } ], "methodology": "Review of documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 87, @@ -405,7 +405,7 @@ } ], "methodology": "Review of technical documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -419,7 +419,7 @@ } ], "methodology": "Review of open-source safety systems", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency as open-source model. Good training data disclosure. Customizable guardrails for specific use cases." @@ -439,7 +439,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 86, @@ -453,7 +453,7 @@ } ], "methodology": "Review of official and community SDKs", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -467,7 +467,7 @@ } ], "methodology": "Review of versioning approach", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 78, @@ -481,7 +481,7 @@ } ], "methodology": "Review of available monitoring tools", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Requires custom monitoring implementation" }, "support_quality": { @@ -502,7 +502,7 @@ } ], "methodology": "Assessment of support channels", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "No model-specific support exists since the model never shipped" }, "ecosystem_maturity": { @@ -520,10 +520,16 @@ "url": "https://en.wikipedia.org/wiki/Llama_(language_model)", "date": "2026-06-10", "value": "No ecosystem exists for Behemoth itself; the model was never released and Meta has pivoted to closed-weight models (Muse Spark, April 2026)" + }, + { + "source": "AI CERTs News - Meta Behemoth Cancel Claim", + "url": "https://www.aicerts.ai/news/meta-behemoth-cancel-claim-inside-llama-4s-uncertain-future/", + "date": "2026-07-09", + "value": "Re-verified July 2026: Behemoth remains unreleased and effectively shelved (never formally cancelled); reported root causes were mid-training MoE-routing changes and chunked-attention issues at 2T scale" } ], "methodology": "Analysis of ecosystem", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -537,7 +543,7 @@ } ], "methodology": "Review of license terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Operational scores are largely theoretical: the model was never released, so no deployment, support, or ecosystem exists for it. Meta shipped the closed-weight Muse Spark (April 2026) instead." @@ -637,7 +643,7 @@ "Requires expertise to deploy and maintain", "No managed API service from Meta", "Large model size requires substantial compute resources", - "Never released: still announced-only as of June 2026; Meta gave no update in January 2026 and pivoted to the closed-weight Muse Spark (April 2026), so weights are unavailable" + "Never released: still announced-only as of July 2026; Meta gave no update in January 2026 and pivoted to the closed-weight Muse Spark (April 8, 2026), so weights are unavailable" ], "best_for": [ "Enterprises requiring data sovereignty and on-premise deployment", diff --git a/data/models/llama-4-maverick.json b/data/models/llama-4-maverick.json index 5f70146..51b1bb9 100644 --- a/data/models/llama-4-maverick.json +++ b/data/models/llama-4-maverick.json @@ -4,9 +4,9 @@ "name": "Llama 4 Maverick", "provider": "Meta", "version": "400B", - "last_evaluated": "2025-11-07", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Meta's flagship open-source model with 400B parameters, native multimodal capabilities, and state-of-the-art performance. Best-in-class open-source option for on-premises deployment.", + "description": "Meta's flagship open-weight model (released April 5, 2025): a mixture-of-experts with 400B total / 17B active parameters (128 experts), natively multimodal (text + image), with a 1M-token context window. Meta's last open-weight release: the company has since pivoted to closed models with Muse Spark (April 2026), and newer open models from DeepSeek, Qwen, and Moonshot have surpassed it on most benchmarks.", "website": "https://llama.meta.com/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Standard coding benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 90, @@ -38,7 +38,7 @@ } ], "methodology": "PhD-level reasoning benchmarks", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 92, @@ -52,7 +52,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 89, @@ -66,7 +66,7 @@ } ], "methodology": "Community evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "Depends on hardware", @@ -80,7 +80,7 @@ } ], "methodology": "Community deployment reports", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Self-hosted so latency depends entirely on hardware and optimization" }, "latency_p95": { @@ -95,21 +95,27 @@ } ], "methodology": "Community reports", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "context_window": { - "value": "128,000 tokens", + "value": "1,000,000 tokens", "confidence": "high", "evidence": [ { - "source": "Llama 4 Documentation", - "url": "https://llama.meta.com/docs/", - "date": "2025-07-15", - "value": "128K token context window" + "source": "Meta AI Blog - The Llama 4 herd", + "url": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/", + "date": "2025-04-05", + "value": "1M-token context window for Maverick (Instruct)" + }, + { + "source": "Hugging Face - Welcome Llama 4 Maverick & Scout", + "url": "https://huggingface.co/blog/llama4-release", + "date": "2026-07-09", + "value": "1M context confirmed; hosted providers often expose less" } ], "methodology": "Official specification", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uptime": { "value": "Self-hosted", @@ -123,7 +129,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Uptime is under customer control" } }, @@ -144,7 +150,7 @@ } ], "methodology": "Community security testing", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Open-source nature enables security hardening but requires customer implementation" }, "jailbreak_resistance": { @@ -159,7 +165,7 @@ } ], "methodology": "Safety evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 95, @@ -173,7 +179,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07", + "last_verified": "2026-07-09", "notes": "Self-hosting eliminates external data leakage risk" }, "output_safety": { @@ -188,7 +194,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "api_security": { "value": "Customer controlled", @@ -202,7 +208,7 @@ } ], "methodology": "Architecture analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Security is customer-controlled with self-hosting. Excellent for data sovereignty but requires in-house security expertise." @@ -222,7 +228,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 100, @@ -236,7 +242,7 @@ } ], "methodology": "Deployment architecture", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Customer controlled", @@ -250,7 +256,7 @@ } ], "methodology": "Architecture analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 95, @@ -264,7 +270,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 90, @@ -278,7 +284,7 @@ } ], "methodology": "Deployment model analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 100, @@ -292,7 +298,7 @@ } ], "methodology": "Architecture analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy - best-in-class. Self-hosting provides complete data control, enabling any compliance framework." @@ -312,7 +318,7 @@ } ], "methodology": "Capability evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 84, @@ -326,7 +332,7 @@ } ], "methodology": "Community evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 82, @@ -340,7 +346,7 @@ } ], "methodology": "Bias benchmark evaluation", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 85, @@ -354,7 +360,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 95, @@ -368,7 +374,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 92, @@ -382,7 +388,7 @@ } ], "methodology": "Research paper review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "guardrails": { "score": 78, @@ -396,7 +402,7 @@ } ], "methodology": "Safety mechanism review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Excellent transparency as open-source model. Full access to weights and detailed documentation." @@ -416,7 +422,7 @@ } ], "methodology": "Tooling review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 88, @@ -430,7 +436,7 @@ } ], "methodology": "SDK ecosystem review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 90, @@ -444,7 +450,7 @@ } ], "methodology": "Release policy review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 80, @@ -458,7 +464,7 @@ } ], "methodology": "Tooling analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "support_quality": { "score": 82, @@ -472,7 +478,7 @@ } ], "methodology": "Support channel assessment", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 92, @@ -486,7 +492,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" }, "license_terms": { "score": 95, @@ -500,7 +506,7 @@ } ], "methodology": "License review", - "last_verified": "2025-11-07" + "last_verified": "2026-07-09" } }, "notes": "Strong operational maturity with massive open-source ecosystem. Requires in-house ML ops expertise." @@ -598,10 +604,11 @@ "limitations": [ "Requires significant ML ops expertise and infrastructure", "Performance and latency depend on hardware investment", - "Slightly behind frontier proprietary models on benchmarks", + "Behind frontier proprietary models and newer open models (DeepSeek V4, Qwen3.5, Kimi K2.6) on benchmarks", "No managed service or enterprise support from Meta", "Requires customer implementation of safety guardrails", - "High upfront hardware costs (8x A100/H100 GPUs minimum)" + "High upfront hardware costs (8x A100/H100 GPUs minimum)", + "Legacy status: Meta's last open-weight release; Meta pivoted to closed models (Muse Spark, April 2026), so no successor open weights are expected" ], "best_for": [ "Highly regulated industries requiring on-premises deployment", @@ -620,9 +627,10 @@ "pricing": { "input": "$0 (self-hosted)", "output": "$0 (self-hosted)", - "notes": "Free model, infrastructure costs only (8x H100 GPUs ~$200K+). No API fees." + "notes": "Free open weights, infrastructure costs only (8x H100 GPUs ~$200K+) for self-hosting. Also served by third-party APIs at low per-token rates as of July 2026.", + "last_verified": "2026-07-09" }, - "context_window": 128000, + "context_window": 1000000, "languages": [ "English", "Spanish", @@ -639,13 +647,12 @@ ], "modalities": [ "text", - "vision", - "audio" + "image (input)" ], - "api_endpoint": "Self-hosted", + "api_endpoint": "Self-hosted or third-party hosts (Together, etc.)", "open_source": true, - "architecture": "Transformer-based, 400B parameters, mixture-of-experts", - "parameters": "400B total, ~60B active" + "architecture": "Mixture-of-experts (128 routed experts + shared expert), natively multimodal via early fusion", + "parameters": "400B total / 17B active" }, "related_entities": [ "claude-opus-4-1", diff --git a/data/models/llama-4-scout.json b/data/models/llama-4-scout.json index 06006b9..8e4c177 100644 --- a/data/models/llama-4-scout.json +++ b/data/models/llama-4-scout.json @@ -3,10 +3,10 @@ "type": "model", "name": "Llama 4 Scout", "provider": "Meta", - "version": "2025-02", - "last_evaluated": "2025-11-08", + "version": "2025-04", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Meta's efficient Llama 4 model optimized for speed and resource efficiency. Designed for edge deployment and cost-sensitive applications requiring open-source flexibility.", + "description": "Meta's efficient Llama 4 model (released April 5, 2025): a natively multimodal mixture-of-experts with 109B total / 17B active parameters (16 experts) and an industry-leading 10M-token context window, deployable on a single H100-class GPU. Optimized for speed and cost-sensitive applications requiring open-weight flexibility. Now a legacy line: Meta has shipped no new open weights since Scout/Maverick and pivoted to closed models with Muse Spark (April 2026).", "website": "https://llama.meta.com/llama-4", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 74, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 77, @@ -52,7 +52,7 @@ } ], "methodology": "Knowledge testing benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 75, @@ -66,7 +66,7 @@ } ], "methodology": "Internal testing with repeated prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "0.6s", @@ -80,7 +80,7 @@ } ], "methodology": "Median latency on recommended hardware", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "1.2s", @@ -94,21 +94,28 @@ } ], "methodology": "95th percentile response time", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { - "value": "64,000 tokens", + "value": "10,000,000 tokens", "confidence": "high", "evidence": [ { - "source": "Meta Documentation", - "url": "https://llama.meta.com/docs/model-cards-and-prompt-formats/llama4-scout", - "date": "2025-02-01", - "value": "64K token context window" + "source": "Meta AI Blog - The Llama 4 herd", + "url": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/", + "date": "2025-04-05", + "value": "Industry-leading 10M-token context window" + }, + { + "source": "Hugging Face - meta-llama/Llama-4-Scout-17B-16E", + "url": "https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E", + "date": "2026-07-09", + "value": "10M context confirmed on model card; hosted API providers typically expose far less (128K-1M)" } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09", + "notes": "10M is the model's native specification; most hosted endpoints cap context much lower" }, "uptime": { "score": 95, @@ -122,7 +129,7 @@ } ], "methodology": "User-controlled deployment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Efficient performance optimized for speed and resource usage. Good balance for edge deployment and cost-sensitive applications." @@ -142,7 +149,7 @@ } ], "methodology": "Testing against prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 79, @@ -156,7 +163,7 @@ } ], "methodology": "Testing against adversarial prompts", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -170,7 +177,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 80, @@ -184,7 +191,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 82, @@ -198,7 +205,7 @@ } ], "methodology": "Review of deployment practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good baseline security with self-hosted deployment providing full control. Smaller model may have slightly lower resistance than Behemoth." @@ -218,7 +225,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 98, @@ -232,7 +239,7 @@ } ], "methodology": "Analysis of data flow", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "User-controlled", @@ -246,7 +253,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 92, @@ -260,7 +267,7 @@ } ], "methodology": "Review of deployment architecture", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 94, @@ -274,7 +281,7 @@ } ], "methodology": "Review of deployment options", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 98, @@ -288,7 +295,7 @@ } ], "methodology": "Analysis of deployment model", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy with self-hosted deployment. Full control over all data aspects." @@ -308,7 +315,7 @@ } ], "methodology": "Evaluation of reasoning transparency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -322,7 +329,7 @@ } ], "methodology": "Community evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 81, @@ -336,7 +343,7 @@ } ], "methodology": "Evaluation on bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 83, @@ -350,7 +357,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 90, @@ -364,7 +371,7 @@ } ], "methodology": "Review of documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 87, @@ -378,7 +385,7 @@ } ], "methodology": "Review of technical documentation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 88, @@ -392,7 +399,7 @@ } ], "methodology": "Review of safety systems", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong transparency as open-source model. Good documentation and customizable guardrails." @@ -412,7 +419,7 @@ } ], "methodology": "Review of API design", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 86, @@ -426,7 +433,7 @@ } ], "methodology": "Review of SDKs", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 88, @@ -440,7 +447,7 @@ } ], "methodology": "Review of versioning", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 80, @@ -454,7 +461,7 @@ } ], "methodology": "Review of monitoring tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 84, @@ -468,7 +475,7 @@ } ], "methodology": "Assessment of support", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 89, @@ -482,7 +489,7 @@ } ], "methodology": "Analysis of ecosystem", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -496,7 +503,7 @@ } ], "methodology": "Review of license", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good operational maturity with strong ecosystem. Easier to deploy than Behemoth due to smaller size." @@ -507,7 +514,7 @@ "overall": 74, "notes": "Adequate for basic coding tasks. Fast inference makes it suitable for development tools.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "gpt-4-1-mini" ] }, @@ -516,7 +523,7 @@ "notes": "Well-suited for customer support with fast response times and privacy benefits.", "alternatives": [ "gpt-4-1-mini", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "content-creation": { @@ -524,14 +531,14 @@ "notes": "Good for content creation with balanced quality and speed.", "alternatives": [ "gpt-4-1", - "llama-4-behemoth" + "llama-4-maverick" ] }, "data-analysis": { "overall": 76, "notes": "Adequate for basic data analysis. Not suitable for complex mathematical tasks.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "gpt-4-1" ] }, @@ -539,7 +546,7 @@ "overall": 77, "notes": "Good for basic research tasks. 57.2% MMLU shows solid general knowledge.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "gpt-4-1" ] }, @@ -547,7 +554,7 @@ "overall": 80, "notes": "Good for basic legal tasks with data sovereignty benefits.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "claude-sonnet-4-5" ] }, @@ -555,7 +562,7 @@ "overall": 84, "notes": "Good for healthcare with self-hosted HIPAA compliance. Basic clinical tasks.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "claude-sonnet-4-5" ] }, @@ -563,7 +570,7 @@ "overall": 75, "notes": "Adequate for basic financial tasks. Not suitable for complex modeling.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "openai-o3" ] }, @@ -571,7 +578,7 @@ "overall": 79, "notes": "Good for educational content. Fast inference suitable for interactive learning.", "alternatives": [ - "llama-4-behemoth", + "llama-4-maverick", "gpt-4-1" ] }, @@ -580,48 +587,50 @@ "notes": "Adequate creative writing for typical use cases.", "alternatives": [ "gpt-4-1", - "llama-4-behemoth" + "llama-4-maverick" ] } }, "strengths": [ "Fast inference (~0.6s p50) suitable for real-time applications", - "Lower resource requirements enable edge deployment", - "Complete data sovereignty with self-hosted deployment", - "Open-source with full transparency", - "No data retention or sharing concerns", + "Fits on a single H100-class GPU (17B active parameters)", + "Industry-leading 10M-token native context window", + "Natively multimodal (text + image input) via early fusion", + "Complete data sovereignty with self-hosted deployment — no data retention or sharing concerns", + "Open weights with full transparency", "Cost-effective for high-volume workloads" ], "limitations": [ - "Moderate accuracy (57.2% MMLU) compared to larger models", - "Limited coding capabilities (42% HumanEval estimated)", - "Smaller context window (64K tokens)", + "Moderate accuracy compared to larger models (Maverick, frontier proprietary)", + "Limited coding capabilities relative to coding-specialized models", + "Native 10M context rarely exposed by hosted providers (typically capped at 128K-1M)", "Requires infrastructure for deployment", "Less capable for complex reasoning tasks", - "No managed API service from Meta" + "No managed API service from Meta", + "Legacy status: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models (Muse Spark, April 2026)" ], "best_for": [ - "Edge deployment with resource constraints", - "High-volume, cost-sensitive applications", + "Single-GPU (H100-class) deployment with resource constraints", + "High-volume, cost-sensitive applications needing fast inference", "Customer support chatbots requiring privacy", - "Development teams prioritizing open-source", - "Applications requiring fast inference" + "Very-long-context document processing when self-hosted (up to 10M tokens)", + "Development teams prioritizing open-source" ], "not_recommended_for": [ "Complex coding or software engineering", "Advanced mathematical or scientific research", "Applications requiring highest accuracy", "Teams without deployment expertise", - "Large document processing (limited context)", "Mission-critical applications" ], "metadata": { "pricing": { "input": "Self-hosted (infrastructure costs)", "output": "Self-hosted (infrastructure costs)", - "notes": "Open-source model. Typically $0.10-0.50 per 1M tokens with optimized deployment." + "notes": "Open-weight model. Typically $0.10-0.50 per 1M tokens via optimized deployment or third-party hosts as of July 2026.", + "last_verified": "2026-07-09" }, - "context_window": 64000, + "context_window": 10000000, "languages": [ "English", "Spanish", @@ -635,15 +644,16 @@ "100+ languages" ], "modalities": [ - "text" + "text", + "image (input)" ], - "api_endpoint": "Self-hosted", + "api_endpoint": "Self-hosted or third-party hosts (Together, Groq, etc.)", "open_source": true, - "architecture": "Transformer-based, optimized for efficiency", - "parameters": "8B (estimated)" + "architecture": "Mixture-of-experts (16 experts), natively multimodal via early fusion", + "parameters": "109B total / 17B active" }, "related_entities": [ - "llama-4-behemoth", + "llama-4-maverick", "llama-3-3-70b", "gpt-4-1-mini", "claude-haiku-4-5" diff --git a/data/models/minimax-m2.json b/data/models/minimax-m2.json index b5541e7..3a799bd 100644 --- a/data/models/minimax-m2.json +++ b/data/models/minimax-m2.json @@ -4,7 +4,7 @@ "name": "MiniMax-M2", "provider": "MiniMax", "version": "20251027", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "MiniMax's MIT-licensed 230B MoE with only 10B active parameters, optimized for agentic tool calling and coding. Topped open-model agentic rankings at launch and undercut Claude pricing by roughly 92% while remaining fast due to its small active footprint.", "website": "https://www.minimax.io/news/minimax-m2", @@ -31,7 +31,7 @@ } ], "methodology": "Vendor benchmarks corroborated by independent press and leaderboard coverage; superseded at the top by 2026 releases", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 85, @@ -45,7 +45,7 @@ } ], "methodology": "Vendor-reported reasoning benchmarks and community evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 84, @@ -59,7 +59,7 @@ } ], "methodology": "Independent composite benchmarking across knowledge domains", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 84, @@ -73,7 +73,7 @@ } ], "methodology": "Community testing of repeated runs and agentic trajectories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.8s", @@ -87,7 +87,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "4.0s", @@ -101,7 +101,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "204,800 tokens", @@ -115,7 +115,7 @@ } ], "methodology": "Official specification from model card", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 94, @@ -129,10 +129,10 @@ } ], "methodology": "Review of platform availability and self-hosting fallback options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Was the leading open agentic model at its October 2025 launch; still strong, but 2026 releases (GLM-5, Kimi K2.6) have surpassed it on raw benchmarks. Its 10B-active design remains a standout for speed and serving cost. Successor MiniMax-M3 announced 2026-06-01 (1M context) but weights not yet published." + "notes": "Was the leading open agentic model at its October 2025 launch; still strong, but 2026 releases (GLM-5, Kimi K2.6) have surpassed it on raw benchmarks. Its 10B-active design remains a standout for speed and serving cost. Successor MiniMax-M3 (428B total / 23B active, 1M context, native multimodality) launched 2026-06-01 with open weights published on Hugging Face by 2026-06-07." }, "security": { @@ -150,7 +150,7 @@ } ], "methodology": "Review of vendor documentation and community testing against OWASP LLM01 patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 75, @@ -164,7 +164,7 @@ } ], "methodology": "Testing against adversarial prompt datasets; deployer-dependent for self-hosted use", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 74, @@ -178,7 +178,7 @@ } ], "methodology": "Analysis of privacy policies and self-hosting data-control options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 78, @@ -192,7 +192,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 82, @@ -206,7 +206,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standard open-model posture without third-party audits. Self-hosting shifts security responsibility to the deployer." @@ -233,7 +233,7 @@ } ], "methodology": "Review of provider jurisdiction and third-party hosting options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 74, @@ -247,7 +247,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Per MiniMax policy on first-party API (China jurisdiction); zero when self-hosted", @@ -261,7 +261,7 @@ } ], "methodology": "Review of terms of service and deployment-dependent retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 70, @@ -275,7 +275,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 66, @@ -289,7 +289,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 85, @@ -303,7 +303,7 @@ } ], "methodology": "Review of self-hosting deployment options enabling zero retention", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "First-party MiniMax API operates under Chinese jurisdiction — a material caveat for Western regulated industries. The small 10B-active footprint makes self-hosted mitigation cheaper than for other frontier-scale open models." @@ -324,7 +324,7 @@ } ], "methodology": "Evaluation of reasoning transparency and trajectory inspectability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 78, @@ -338,7 +338,7 @@ } ], "methodology": "Testing on factual QA datasets and tool-augmented workflows", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 74, @@ -352,7 +352,7 @@ } ], "methodology": "Review of published bias benchmarks and community evaluations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 76, @@ -366,7 +366,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 85, @@ -380,7 +380,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 72, @@ -394,7 +394,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 76, @@ -408,7 +408,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Open weights and interleaved-thinking traces provide reasonable transparency; training data disclosure and formal bias/safety evaluations are limited." @@ -429,7 +429,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 80, @@ -443,21 +443,21 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 76, "confidence": "medium", "evidence": [ { - "source": "MiniMax-M3 announcement", - "url": "https://www.minimax.io/news", - "date": "2026-06-01", - "value": "Successor M3 announced 2026-06-01 with 1M context, but weights not yet published as of 2026-06-10; M2 weights remain available" + "source": "MiniMax-M3 announcement and weights release", + "url": "https://www.minimax.io/blog/minimax-m3", + "date": "2026-07-09", + "value": "Successor M3 (428B/23B, 1M context, native multimodal) launched 2026-06-01; open weights published on Hugging Face by 2026-06-07, though training code and inference operators were not released. M2 weights remain available" } ], "methodology": "Review of versioning practices and weight availability across releases", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 76, @@ -471,7 +471,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 76, @@ -485,7 +485,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 84, @@ -499,7 +499,7 @@ } ], "methodology": "Analysis of third-party hosting, integrations, and tooling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 94, @@ -513,10 +513,10 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Clean MIT licensing and dual OpenAI/Anthropic API compatibility lower switching costs. M3 transition (announced, weights unpublished) is the main forward-looking uncertainty." + "notes": "Clean MIT licensing and dual OpenAI/Anthropic API compatibility lower switching costs. Successor M3's weights shipped 2026-06-07, resolving the earlier roadmap uncertainty; M2 remains available but is now the previous generation." } }, @@ -585,7 +585,7 @@ "limitations": [ "First-party MiniMax API processes data under Chinese jurisdiction with no published Western compliance certifications", "Surpassed on raw benchmarks by 2026 open-weight releases (GLM-5, Kimi K2.6)", - "Successor M3 announced (2026-06-01) but weights unpublished, creating roadmap uncertainty", + "Superseded within MiniMax's own lineup: M3 (1M context, native multimodality) shipped with open weights in June 2026", "Text-only — no vision or audio modalities", "Limited published bias, safety, and red-team evaluations", "Interleaved thinking format requires prompt-handling care in some frameworks" @@ -608,8 +608,8 @@ "pricing": { "input": "$0.30 per 1M tokens (approx.)", "output": "$1.20 per 1M tokens (approx.)", - "notes": "Launched at roughly 8% of Claude Sonnet pricing; third-party host pricing varies.", - "last_verified": "2026-06-10" + "notes": "Launched at roughly 8% of Claude Sonnet pricing; third-party host pricing varies. First-party rates unchanged in July 2026 checks.", + "last_verified": "2026-07-09" }, "context_window": 204800, "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], diff --git a/data/models/mistral-large-3.json b/data/models/mistral-large-3.json index 3b9d97e..427ac4b 100644 --- a/data/models/mistral-large-3.json +++ b/data/models/mistral-large-3.json @@ -4,7 +4,7 @@ "name": "Mistral Large 3", "provider": "Mistral AI", "version": "Large 3 (Mistral 3 family)", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Mistral AI's open-weight flagship released December 2025 under Apache 2.0: a sparse MoE (675B total / 41B active) multimodal model with ~256K context and 40+ languages. Debuted #2 among open-source non-reasoning models on LMArena, with a strong EU data-sovereignty story.", "website": "https://mistral.ai/news/mistral-3/", @@ -24,7 +24,7 @@ } ], "methodology": "Review of provider benchmarks and community evaluations of open weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 86, @@ -38,7 +38,7 @@ } ], "methodology": "Review of reasoning benchmarks; model is non-reasoning class (no extended thinking)", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 88, @@ -58,7 +58,7 @@ } ], "methodology": "Crowdsourced arena comparisons and provider benchmark suite", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 85, @@ -72,7 +72,7 @@ } ], "methodology": "Community repeated-prompt testing on open weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "~1.5s (hosted API); deployment-dependent when self-hosted", @@ -86,7 +86,7 @@ } ], "methodology": "Median latency from third-party benchmarking of hosted endpoints", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "~3.5s (hosted API)", @@ -100,7 +100,7 @@ } ], "methodology": "95th percentile estimates across hosting providers", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "~256,000 tokens", @@ -114,7 +114,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 94, @@ -128,7 +128,7 @@ } ], "methodology": "Historical uptime of hosted API; open weights enable customer-controlled availability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Best-in-class open-weight performance for its release window: sparse MoE (675B total / 41B active) delivers near-frontier quality with modest active compute. Non-reasoning class — frontier reasoning models outperform it on hard multi-step problems." @@ -148,7 +148,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection patterns", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 82, @@ -162,7 +162,7 @@ } ], "methodology": "Adversarial prompt testing on hosted and open-weight deployments", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Open weights mean downstream deployments can weaken or strengthen safety behavior" }, "data_leakage_prevention": { @@ -177,7 +177,7 @@ } ], "methodology": "Analysis of privacy policies plus self-hosting option", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 83, @@ -191,7 +191,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -205,7 +205,7 @@ } ], "methodology": "Review of API security features across hosting options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Solid security with the open-weights caveat: deployers control (and can remove) guardrails, so deployment-level controls matter more than for closed models." @@ -225,7 +225,7 @@ } ], "methodology": "Review of hosting documentation and deployment options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -239,7 +239,7 @@ } ], "methodology": "Analysis of terms of service and data usage policy", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Limited retention on hosted API; zero when self-hosted", @@ -253,7 +253,7 @@ } ], "methodology": "Review of retention policies across deployment modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -267,7 +267,7 @@ } ], "methodology": "Review of data protection capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -281,7 +281,7 @@ } ], "methodology": "Verification of certifications across Mistral and cloud hosting partners", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 88, @@ -295,7 +295,7 @@ } ], "methodology": "Review of self-hosting options enabling complete data control", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Standout data-sovereignty story: EU provider under GDPR, plus Apache 2.0 weights allow fully on-premises/air-gapped deployment — the strongest possible residency guarantee." @@ -315,7 +315,7 @@ } ], "methodology": "Evaluation of reasoning transparency; open weights enable interpretability research", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -329,7 +329,7 @@ } ], "methodology": "Factual QA testing by community evaluators", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 78, @@ -343,7 +343,7 @@ } ], "methodology": "Review of bias disclosures and multilingual evaluation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 78, @@ -357,7 +357,7 @@ } ], "methodology": "Qualitative assessment plus open-weight logprob access", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 88, @@ -371,7 +371,7 @@ } ], "methodology": "Review of published model card and architecture disclosure", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 75, @@ -385,7 +385,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 78, @@ -399,7 +399,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Open weights provide architectural transparency rare at this scale (675B MoE disclosed), though training data detail and built-in guardrails are lighter than closed frontier models." @@ -419,7 +419,7 @@ } ], "methodology": "Review of API design and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 84, @@ -433,7 +433,7 @@ } ], "methodology": "Review of SDK and inference-stack support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 84, @@ -447,7 +447,7 @@ } ], "methodology": "Review of versioning policy; open weights eliminate forced-retirement risk", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 80, @@ -461,7 +461,7 @@ } ], "methodology": "Review of monitoring tools across deployment modes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 82, @@ -475,7 +475,7 @@ } ], "methodology": "Assessment of documentation and support channels", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 88, @@ -489,7 +489,7 @@ } ], "methodology": "Analysis of distribution channels and third-party tooling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 96, @@ -503,7 +503,7 @@ } ], "methodology": "Review of license terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Apache 2.0 licensing at frontier scale is the headline: no usage restrictions, no vendor lock-in, and availability across HF, Bedrock, Azure, and La Plateforme." @@ -572,7 +572,7 @@ "limitations": [ "Non-reasoning class — trails frontier reasoning models on hard multi-step problems", "Self-hosting 675B weights requires substantial GPU infrastructure despite 41B active", - "API pricing from aggregators (~$0.50/$1.50) carries medium confidence", + "Hosted API pricing ($2/$6 per 1M on La Plateforme) is several times higher than early aggregator estimates suggested, though output remains cheaper than closed flagships", "Open weights let deployers strip guardrails, shifting safety burden downstream", "Training data composition only described at a high level" ], @@ -589,10 +589,10 @@ ], "metadata": { "pricing": { - "input": "$0.50 per 1M tokens (approx.)", - "output": "$1.50 per 1M tokens (approx.)", - "notes": "Aggregator-reported API pricing, medium confidence; varies by host (La Plateforme, Bedrock, Azure). Self-hosting under Apache 2.0 incurs only infrastructure cost.", - "last_verified": "2026-06-10" + "input": "$2.00 per 1M tokens", + "output": "$6.00 per 1M tokens", + "notes": "Official La Plateforme pricing per mistral.ai/pricing (verified 2026-07-09); corrects earlier aggregator-based estimate of ~$0.50/$1.50. Batch processing gets a 50% discount; cloud-host (Bedrock/Azure) rates vary. Self-hosting under Apache 2.0 incurs only infrastructure cost.", + "last_verified": "2026-07-09" }, "context_window": 256000, "languages": [ diff --git a/data/models/nemotron-ultra-253b.json b/data/models/nemotron-ultra-253b.json index 754c10b..19ab86f 100644 --- a/data/models/nemotron-ultra-253b.json +++ b/data/models/nemotron-ultra-253b.json @@ -4,9 +4,9 @@ "name": "Nemotron Ultra 253B", "provider": "NVIDIA", "version": "20251101", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Massive 253B parameter model from NVIDIA's Llama-3.1-based Nemotron line, now superseded: NVIDIA discontinued this line in favor of the native Nemotron 3 family (announced 2025-12-15). Historically achieved 57.1% on SWE-bench and 80.08% on HumanEval, optimized for high-performance computing and complex coding tasks with GPU acceleration. New deployments should evaluate Nemotron 3 instead.", + "description": "253B parameter model from NVIDIA's Llama-3.1-based Nemotron line, now superseded: NVIDIA discontinued this line in favor of the native Nemotron 3 family, whose rollout completed in June 2026 (Nano 2025-12, Super 2026-03, and Nemotron 3 Ultra — a 550B total / 55B active MoE hybrid Mamba-Transformer — on 2026-06-04). Historically 57.1% SWE-bench and 80.08% HumanEval, optimized for HPC and complex coding with GPU acceleration. New deployments should evaluate Nemotron 3 Ultra instead.", "website": "https://www.nvidia.com/en-us/ai/nemotron/", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring real-world software engineering tasks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 91, @@ -50,7 +50,7 @@ } ], "methodology": "Graduate-level reasoning benchmarks requiring multi-step problem solving", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 90, @@ -70,7 +70,7 @@ } ], "methodology": "Comprehensive knowledge testing across domains", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 89, @@ -84,7 +84,7 @@ } ], "methodology": "Internal testing with repeated prompts at various temperature settings", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Large model size provides good stability" }, "latency_p50": { @@ -99,7 +99,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.8s", @@ -113,7 +113,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -127,7 +127,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -141,7 +141,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent performance for a 253B parameter model with strong coding capabilities. GPU acceleration provides competitive latency despite model size." @@ -161,7 +161,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 86, @@ -175,7 +175,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 82, @@ -189,7 +189,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 88, @@ -203,7 +203,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 87, @@ -217,7 +217,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Solid security posture with enterprise-grade guardrails. Good protection for typical use cases." @@ -237,7 +237,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -251,7 +251,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days (configurable)", @@ -265,7 +265,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 84, @@ -279,7 +279,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 89, @@ -293,7 +293,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 85, @@ -307,7 +307,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good privacy posture with enterprise options. SOC 2 Type II certified with configurable data retention." @@ -327,7 +327,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 82, @@ -341,7 +341,7 @@ } ], "methodology": "Testing on factual QA datasets and real-world usage", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 80, @@ -355,7 +355,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 81, @@ -369,7 +369,7 @@ } ], "methodology": "Assessment of confidence expression in outputs", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 87, @@ -383,7 +383,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 76, @@ -397,7 +397,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 88, @@ -411,7 +411,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with comprehensive documentation. Standard hallucination and bias performance for models of this size." @@ -431,7 +431,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 91, @@ -445,7 +445,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 78, @@ -462,11 +462,17 @@ "url": "https://nvidianews.nvidia.com/news/nvidia-debuts-nemotron-3-family-of-open-models", "date": "2026-06-10", "value": "Llama-3.1-based Nemotron line discontinued in favor of the native Nemotron 3 family (announced 2025-12-15)" + }, + { + "source": "NVIDIA Nemotron 3 Ultra release", + "url": "https://research.nvidia.com/labs/nemotron/Nemotron-3-Ultra/", + "date": "2026-07-09", + "value": "Nemotron 3 rollout complete: Nemotron 3 Ultra (550B total / 55B active MoE hybrid Mamba-Transformer, open weights) released 2026-06-04 as the direct successor to this model" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2026-06-10", - "notes": "Model line discontinued; future development is in the Nemotron 3 family" + "last_verified": "2026-07-09", + "notes": "Model line discontinued; the replacement Nemotron 3 family is now fully shipped (Nano, Super, Ultra)" }, "monitoring_observability": { "score": 88, @@ -480,7 +486,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 89, @@ -494,7 +500,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 88, @@ -508,7 +514,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -522,7 +528,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent operational maturity leveraging NVIDIA's GPU ecosystem. Strong support and comprehensive monitoring tools." @@ -533,7 +539,7 @@ "overall": 93, "notes": "Excellent coding with 57.1% SWE-bench and 80.08% HumanEval. Strong performance on GPU-accelerated workloads.", "alternatives": [ - "claude-4-sonnet", + "claude-sonnet-4", "openai-o1" ] }, @@ -541,23 +547,21 @@ "overall": 84, "notes": "Good conversational capabilities but not specialized for customer support scenarios.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "content-creation": { "overall": 86, "notes": "Solid content generation capabilities with good structure.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "data-analysis": { "overall": 91, "notes": "Strong analytical capabilities, especially for GPU-accelerated data processing.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -565,7 +569,7 @@ "overall": 88, "notes": "Good research capabilities with comprehensive knowledge base.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -573,21 +577,21 @@ "overall": 85, "notes": "Good privacy posture with SOC 2 Type II. Enterprise options for compliance.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "healthcare": { "overall": 83, "notes": "SOC 2 certified but not HIPAA eligible by default. Enterprise options may be available.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "financial-analysis": { "overall": 89, "notes": "Strong analytical and mathematical capabilities.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "openai-o1" ] }, @@ -595,16 +599,14 @@ "overall": 87, "notes": "Good tutoring capabilities with clear explanations.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "creative-writing": { "overall": 84, "notes": "Competent creative capabilities but not specialized for creative writing.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] } }, @@ -623,7 +625,7 @@ "30-day default data retention (not ephemeral)", "Moderate latency (2.2s p50) despite GPU acceleration", "Smaller context window (128K) compared to competitors", - "Superseded: NVIDIA discontinued the Llama-3.1-based Nemotron line in favor of the native Nemotron 3 family (announced 2025-12-15)" + "Superseded: NVIDIA discontinued the Llama-3.1-based Nemotron line; its successor Nemotron 3 Ultra (550B MoE) shipped 2026-06-04, completing the Nemotron 3 family rollout" ], "best_for": [ "GPU-accelerated workloads and HPC environments", @@ -643,7 +645,7 @@ "pricing": { "input": "$0.60 per 1M tokens", "output": "$1.80 per 1M tokens", - "notes": "Competitive pricing with GPU-optimized inference available", + "notes": "Competitive pricing with GPU-optimized inference available. Pricing for this superseded model not re-verifiable against current NVIDIA pages as of 2026-07-09 — confirm availability and rates before procurement.", "last_verified": "2025-11-09" }, "context_window": 128000, diff --git a/data/models/nova-2-lite.json b/data/models/nova-2-lite.json index f397c09..41811ca 100644 --- a/data/models/nova-2-lite.json +++ b/data/models/nova-2-lite.json @@ -4,7 +4,7 @@ "name": "Amazon Nova 2 Lite", "provider": "Amazon (AWS)", "version": "Nova 2 Lite (GA)", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "Amazon's cost-efficient Nova 2 workhorse model, GA on Amazon Bedrock since re:Invent 2025. Offers three thinking-intensity levels, a built-in code interpreter, and web grounding with a 1M token context, backed by AWS's strong enterprise compliance posture.", "website": "https://aws.amazon.com/bedrock/nova/", @@ -24,7 +24,7 @@ } ], "methodology": "Review of provider claims; limited independent benchmark coverage to date", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Public third-party benchmark data sparse; score weighted toward provider evaluations" }, "task_accuracy_reasoning": { @@ -39,7 +39,7 @@ } ], "methodology": "Review of provider evaluations of adjustable-thinking reasoning performance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 83, @@ -53,7 +53,7 @@ } ], "methodology": "Review of provider documentation; independent leaderboard coverage remains sparse", - "last_verified": "2026-06-10", + "last_verified": "2026-07-09", "notes": "Low confidence pending broader third-party benchmarking" }, "output_consistency": { @@ -68,7 +68,7 @@ } ], "methodology": "Review of feature design and repeated-prompt behavior reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "~1.5s (low thinking intensity)", @@ -82,7 +82,7 @@ } ], "methodology": "Median latency from third-party benchmarking at default settings", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "~4.0s (varies with thinking intensity)", @@ -96,7 +96,7 @@ } ], "methodology": "95th percentile estimates across configurations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "1,000,000 tokens (~65K output)", @@ -110,7 +110,7 @@ } ], "methodology": "Official specification from provider documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 97, @@ -124,10 +124,10 @@ } ], "methodology": "Historical service availability from AWS status reporting", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "GA member of the Nova 2 family (Nova 2 Pro and Omni remain in preview). Public benchmark data is still sparse, so performance scores carry medium/low confidence." + "notes": "GA member of the Nova 2 family (Nova 2 Pro and Omni remain in preview as of 2026-07-09; Nova 2 Sonic and Multimodal Embeddings are also GA). Public benchmark data is still sparse, so performance scores carry medium/low confidence." }, "security": { "overall_score": 88, @@ -144,7 +144,7 @@ } ], "methodology": "Testing against OWASP LLM01 patterns with and without Bedrock Guardrails", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 85, @@ -158,7 +158,7 @@ } ], "methodology": "Review of adversarial testing and platform guardrail capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 92, @@ -172,7 +172,7 @@ } ], "methodology": "Analysis of Bedrock data protection documentation and commitments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 86, @@ -186,7 +186,7 @@ } ], "methodology": "Safety testing across harmful content categories", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 92, @@ -200,7 +200,7 @@ } ], "methodology": "Review of AWS-native security controls available to Bedrock workloads", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Inherits AWS's mature platform security: IAM, PrivateLink, KMS, CloudTrail, and Bedrock Guardrails give it one of the strongest deployment security stories available." @@ -220,7 +220,7 @@ } ], "methodology": "Review of Bedrock regional processing documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -234,7 +234,7 @@ } ], "methodology": "Analysis of Bedrock data usage commitments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Not retained by Bedrock (customer-controlled logging)", @@ -248,7 +248,7 @@ } ], "methodology": "Review of Bedrock retention and logging documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 86, @@ -262,7 +262,7 @@ } ], "methodology": "Review of platform PII detection and redaction capabilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 95, @@ -276,7 +276,7 @@ } ], "methodology": "Verification against AWS services-in-scope compliance listings", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 85, @@ -290,7 +290,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Among the strongest compliance postures in the market via Bedrock: SOC, ISO, HIPAA-eligible, GDPR-supporting, with data never used for training." @@ -310,7 +310,7 @@ } ], "methodology": "Evaluation of reasoning transparency and grounded citation behavior", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 80, @@ -324,7 +324,7 @@ } ], "methodology": "Review of grounding features; independent factual QA coverage sparse", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 82, @@ -338,7 +338,7 @@ } ], "methodology": "Review of AWS responsible AI documentation and service cards", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 78, @@ -352,7 +352,7 @@ } ], "methodology": "Qualitative assessment of confidence expression", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 85, @@ -366,7 +366,7 @@ } ], "methodology": "Review of documentation completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 70, @@ -380,7 +380,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 88, @@ -394,7 +394,7 @@ } ], "methodology": "Analysis of built-in and platform-level safety mechanisms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong platform-level guardrails and grounding features; independent transparency and factuality data still limited for the Nova 2 generation." @@ -414,7 +414,7 @@ } ], "methodology": "Review of API design and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 90, @@ -428,7 +428,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 86, @@ -442,7 +442,7 @@ } ], "methodology": "Review of Bedrock model lifecycle policy", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 92, @@ -456,7 +456,7 @@ } ], "methodology": "Review of monitoring and observability integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 90, @@ -470,7 +470,7 @@ } ], "methodology": "Assessment of support tiers and documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 88, @@ -484,7 +484,7 @@ } ], "methodology": "Analysis of platform integrations and tooling", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 88, @@ -498,7 +498,7 @@ } ], "methodology": "Review of licensing terms and indemnification", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Exceptional operational maturity by virtue of the AWS/Bedrock platform: IAM, monitoring, lifecycle policy, and enterprise support are best-in-class." @@ -587,8 +587,8 @@ "pricing": { "input": "$0.30 per 1M tokens", "output": "$2.50 per 1M tokens", - "notes": "Amazon Bedrock pricing; batch and provisioned throughput options available. Announced at re:Invent 2025-12-02.", - "last_verified": "2026-06-10" + "notes": "Amazon Bedrock pricing; batch (50% discount) and provisioned throughput options available. Announced at re:Invent 2025-12-02; rates re-confirmed unchanged July 2026.", + "last_verified": "2026-07-09" }, "context_window": 1000000, "max_output": 65000, diff --git a/data/models/nova-pro.json b/data/models/nova-pro.json index dc17c2f..57b4a7a 100644 --- a/data/models/nova-pro.json +++ b/data/models/nova-pro.json @@ -4,9 +4,9 @@ "name": "Nova Pro", "provider": "Amazon", "version": "2025-01", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "SUPERSEDED by the Nova 2 family announced at re:Invent 2025-12-02 (Nova 2 Lite GA; Nova 2 Pro/Omni in preview), though the original Nova Pro is still served. Amazon model integrated with AWS services for enterprise customers requiring seamless AWS integration. New projects should evaluate Nova 2.", + "description": "SUPERSEDED by the Nova 2 family announced at re:Invent 2025-12-02 (Nova 2 Lite GA; Nova 2 Pro/Omni still in preview as of 2026-07-09), though the original Nova Pro is still served. Amazon model integrated with AWS services for enterprise customers requiring seamless AWS integration. New projects should evaluate Nova 2.", "website": "https://aws.amazon.com/bedrock/nova/", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 72, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical reasoning", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 74, @@ -52,7 +52,7 @@ } ], "methodology": "Knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 75, @@ -66,7 +66,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.0s", @@ -80,7 +80,7 @@ } ], "methodology": "AWS infrastructure", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "2.0s", @@ -94,7 +94,7 @@ } ], "methodology": "CloudWatch metrics", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "300,000 tokens", @@ -108,7 +108,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -122,7 +122,7 @@ } ], "methodology": "AWS service SLA", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Moderate performance with excellent AWS integration. Large 300K context window. Reliable AWS infrastructure." @@ -142,7 +142,7 @@ } ], "methodology": "AWS security testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -156,7 +156,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 92, @@ -170,7 +170,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 88, @@ -184,7 +184,7 @@ } ], "methodology": "Safety benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 93, @@ -198,7 +198,7 @@ } ], "methodology": "Security architecture review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent security leveraging AWS infrastructure. Strong IAM and VPC integration." @@ -218,7 +218,7 @@ } ], "methodology": "AWS infrastructure review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 95, @@ -232,7 +232,7 @@ } ], "methodology": "Privacy policy", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "0 days (ephemeral)", @@ -246,7 +246,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 90, @@ -260,7 +260,7 @@ } ], "methodology": "Feature review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 95, @@ -274,7 +274,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 95, @@ -288,7 +288,7 @@ } ], "methodology": "Architecture review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Exceptional privacy with AWS infrastructure. HIPAA eligible, FedRAMP authorized. Zero data retention." @@ -308,7 +308,7 @@ } ], "methodology": "Reasoning evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 78, @@ -322,7 +322,7 @@ } ], "methodology": "Factual QA testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 80, @@ -336,7 +336,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 79, @@ -350,7 +350,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 87, @@ -364,7 +364,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 76, @@ -378,7 +378,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -392,7 +392,7 @@ } ], "methodology": "Feature review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with AWS Guardrails. Comprehensive documentation." @@ -412,7 +412,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 96, @@ -426,7 +426,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 92, @@ -441,12 +441,12 @@ { "source": "AWS What's New: Nova 2 foundation models in Amazon Bedrock", "url": "https://aws.amazon.com/about-aws/whats-new/2025/12/nova-2-foundation-models-amazon-bedrock/", - "date": "2026-06-10", - "value": "Nova 2 family announced 2025-12-02 (Nova 2 Lite GA; Pro/Omni preview); original Nova Pro still served" + "date": "2026-07-09", + "value": "Nova 2 family announced 2025-12-02; as of 2026-07-09 Nova 2 Lite, Sonic, and Multimodal Embeddings are GA while Nova 2 Pro and Omni remain in preview; original Nova Pro still served on Bedrock" } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 97, @@ -460,7 +460,7 @@ } ], "methodology": "Observability review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 95, @@ -474,7 +474,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 94, @@ -488,7 +488,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 93, @@ -502,7 +502,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Exceptional operational excellence with full AWS integration. Best-in-class monitoring and support." @@ -617,8 +617,8 @@ "pricing": { "input": "$0.80 per 1M tokens (on-demand), $0.40 per 1M tokens (batch)", "output": "$3.20 per 1M tokens (on-demand), $1.60 per 1M tokens (batch)", - "notes": "AWS Bedrock pricing with 50% batch discount, volume discounts available", - "last_verified": "2025-11-09" + "notes": "AWS Bedrock pricing with 50% batch discount, volume discounts available. Re-confirmed unchanged ($0.80/$3.20 on-demand) as of July 2026.", + "last_verified": "2026-07-09" }, "context_window": 300000, "languages": [ diff --git a/data/models/openai-o1-mini.json b/data/models/openai-o1-mini.json index 3d16d77..a848efe 100644 --- a/data/models/openai-o1-mini.json +++ b/data/models/openai-o1-mini.json @@ -4,9 +4,9 @@ "name": "OpenAI o1-mini", "provider": "OpenAI", "version": "2024-12", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DEPRECATED: o1-mini (with o1-preview) was removed from the OpenAI API in 2025 and from ChatGPT; remaining o1 variants shut down 2026-10-23. Migration target is GPT-5.5. Historically an efficient reasoning model with chain-of-thought capabilities at lower cost than o1.", + "description": "RETIRED: o1-mini was shut down in the OpenAI API on 2025-10-27 (o1-preview on 2025-07-28) and removed from ChatGPT; it is no longer available anywhere. OpenAI's stated replacement was o4-mini, itself now scheduled for shutdown 2026-10-23 — migrate new work to the GPT-5.x family (gpt-5.5 / gpt-5.4-mini). Historically an efficient reasoning model with chain-of-thought capabilities at lower cost than o1.", "website": "https://openai.com/o1", "trust_vector": { "performance_reliability": { @@ -24,7 +24,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 81, @@ -38,7 +38,7 @@ } ], "methodology": "Mathematical reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 81, @@ -52,7 +52,7 @@ } ], "methodology": "Knowledge testing benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 83, @@ -66,7 +66,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.8s", @@ -80,7 +80,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.6s", @@ -94,7 +94,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -108,7 +108,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -122,7 +122,7 @@ } ], "methodology": "Historical uptime", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good reasoning performance with faster inference than o3. Balanced for cost-sensitive reasoning tasks." @@ -142,7 +142,7 @@ } ], "methodology": "OWASP LLM01 testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 87, @@ -156,7 +156,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -170,7 +170,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 86, @@ -184,7 +184,7 @@ } ], "methodology": "Safety benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -198,7 +198,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong security with reasoning-enhanced safety." @@ -218,7 +218,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -232,7 +232,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -246,7 +246,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -260,7 +260,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -274,7 +274,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -288,7 +288,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Standard OpenAI privacy with 30-day retention." @@ -308,7 +308,7 @@ } ], "methodology": "Reasoning transparency evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 85, @@ -322,7 +322,7 @@ } ], "methodology": "Factual QA testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 80, @@ -336,7 +336,7 @@ } ], "methodology": "Bias benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 84, @@ -350,7 +350,7 @@ } ], "methodology": "Qualitative assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 86, @@ -364,7 +364,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -378,7 +378,7 @@ } ], "methodology": "Public disclosure review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 87, @@ -392,7 +392,7 @@ } ], "methodology": "Safety system analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent explainability via chain-of-thought. Good transparency." @@ -412,7 +412,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -426,7 +426,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 73, @@ -441,12 +441,12 @@ { "source": "OpenAI Deprecations", "url": "https://developers.openai.com/api/docs/deprecations", - "date": "2026-06-10", - "value": "o1-preview/o1-mini removed in 2025; remaining o1 variants API shutdown 2026-10-23; migration target GPT-5.5" + "date": "2026-07-09", + "value": "o1-mini shutdown date 2025-10-27 (recommended replacement o4-mini); o1-preview shutdown 2025-07-28 (replacement o3); remaining o1/o1-pro variants shut down 2026-10-23 with GPT-5.5 as migration target" } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 84, @@ -460,7 +460,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 87, @@ -474,7 +474,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 84, @@ -488,7 +488,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -502,10 +502,10 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, - "notes": "Deprecated: o1-mini removed from API in 2025; migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." + "notes": "Retired: o1-mini API shut down 2025-10-27 (per OpenAI deprecations page, verified 2026-07-09). Model is no longer accessible; entry retained for historical reference. Versioning and ecosystem scores reduced to reflect retirement." } }, "use_case_ratings": { @@ -522,7 +522,7 @@ "notes": "Reasoning overhead may be unnecessary for basic support.", "alternatives": [ "gpt-4-1", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "content-creation": { @@ -599,7 +599,7 @@ "Reasoning overhead unnecessary for simple tasks", "Lower performance than o3 on complex tasks", "Premium pricing for reasoning capabilities", - "DEPRECATED: removed from the API in 2025 and from ChatGPT — migrate to GPT-5.5" + "RETIRED: API shut down 2025-10-27; removed from ChatGPT — no longer available, migrate to GPT-5.x (gpt-5.5 / gpt-5.4-mini)" ], "best_for": [ "Reasoning tasks requiring transparency", @@ -620,8 +620,8 @@ "pricing": { "input": "$3.00 per 1M tokens", "output": "$12.00 per 1M tokens", - "notes": "Mid-tier reasoning model pricing (pricing varies by tier and usage)", - "last_verified": "2025-11-09" + "notes": "Historical pricing only — model retired 2025-10-27 and can no longer be purchased", + "last_verified": "2026-07-09" }, "context_window": 128000, "languages": [ @@ -650,7 +650,7 @@ "claude-sonnet-4-5" ], "tags": [ - "deprecated", + "retired", "reasoning", "chain-of-thought", "coding", diff --git a/data/models/openai-o1.json b/data/models/openai-o1.json index 9db339f..1985461 100644 --- a/data/models/openai-o1.json +++ b/data/models/openai-o1.json @@ -4,7 +4,7 @@ "name": "OpenAI o1", "provider": "OpenAI", "version": "20250915", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "DEPRECATED: o1 variants and o1-pro shut down in the API on 2026-10-23; already removed from ChatGPT (o1-preview/o1-mini removed in 2025). Migration target is GPT-5.5. Historically an advanced reasoning model (57.1% SWE-bench, 79.2% HumanEval) with extended chain-of-thought reasoning.", "website": "https://openai.com/o1/", @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring real-world software engineering tasks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -50,7 +50,7 @@ } ], "methodology": "Competition-level reasoning benchmarks requiring extended chain-of-thought", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 91, @@ -70,7 +70,7 @@ } ], "methodology": "Comprehensive knowledge testing across domains", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 92, @@ -84,7 +84,7 @@ } ], "methodology": "Internal testing with repeated prompts at various temperature settings", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Extended reasoning provides more consistent problem-solving" }, "latency_p50": { @@ -99,7 +99,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "8.2s", @@ -113,7 +113,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -127,7 +127,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -141,7 +141,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Exceptional reasoning capabilities with extended chain-of-thought. Best for complex problem-solving requiring deep thinking. Higher latency due to reasoning overhead." @@ -161,7 +161,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 90, @@ -175,7 +175,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 85, @@ -189,7 +189,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 92, @@ -203,7 +203,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 89, @@ -217,7 +217,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong security posture with enhanced reasoning-based safety. Good protection against common attacks." @@ -237,7 +237,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 92, @@ -251,7 +251,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days (minimum for abuse monitoring)", @@ -265,7 +265,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -279,7 +279,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 88, @@ -293,7 +293,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 82, @@ -307,7 +307,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good privacy posture with SOC 2 certification. 30-day minimum retention for safety monitoring." @@ -327,7 +327,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -341,7 +341,7 @@ } ], "methodology": "Testing on factual QA datasets and real-world usage", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 84, @@ -355,7 +355,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 89, @@ -369,7 +369,7 @@ } ], "methodology": "Assessment of confidence expression in outputs", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 92, @@ -383,7 +383,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 80, @@ -397,7 +397,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 93, @@ -411,7 +411,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent explainability via chain-of-thought reasoning. Transparent problem-solving process visible to users." @@ -431,7 +431,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 94, @@ -445,7 +445,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 78, @@ -460,12 +460,12 @@ { "source": "OpenAI Deprecations", "url": "https://developers.openai.com/api/docs/deprecations", - "date": "2026-06-10", - "value": "o1 variants and o1-pro API shutdown 2026-10-23; migration target GPT-5.5; removed from ChatGPT" + "date": "2026-07-09", + "value": "o1 (o1-2024-12-17) API shutdown 2026-10-23, replacement gpt-5.5; o1-pro shutdown 2026-10-23, replacement gpt-5.5-pro; o1-preview shut down 2025-07-28 and o1-mini 2025-10-27" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 89, @@ -479,7 +479,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 90, @@ -493,7 +493,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 83, @@ -507,7 +507,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 91, @@ -521,7 +521,7 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Deprecated: API shutdown scheduled 2026-10-23, migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." @@ -532,31 +532,29 @@ "overall": 93, "notes": "Excellent coding with 57.1% SWE-bench and 79.2% HumanEval. Chain-of-thought helps with complex algorithms.", "alternatives": [ - "claude-4-sonnet", - "claude-4-opus" + "claude-sonnet-4", + "claude-opus-4" ] }, "customer-support": { "overall": 82, "notes": "Good capabilities but high latency (4.5s) may impact customer experience. Better for complex issues.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "content-creation": { "overall": 85, "notes": "Good content generation but reasoning focus may add unnecessary latency for creative tasks.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "data-analysis": { "overall": 95, "notes": "Exceptional analytical capabilities with chain-of-thought reasoning. Best for complex analysis.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "nemotron-ultra-253b" ] }, @@ -564,29 +562,28 @@ "overall": 96, "notes": "Outstanding research capabilities with transparent reasoning. Excellent for complex research tasks.", "alternatives": [ - "claude-4-opus", - "claude-3-7-sonnet-r" + "claude-opus-4" ] }, "legal-compliance": { "overall": 84, "notes": "Good reasoning for legal analysis but 30-day retention may be concern for some use cases.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "healthcare": { "overall": 83, "notes": "Good reasoning but not HIPAA eligible. 30-day retention may be concern for healthcare data.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "financial-analysis": { "overall": 94, "notes": "Outstanding for complex financial modeling and analysis with transparent reasoning.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "nemotron-ultra-253b" ] }, @@ -594,16 +591,14 @@ "overall": 96, "notes": "Exceptional for education with visible chain-of-thought. Perfect for teaching problem-solving.", "alternatives": [ - "claude-4-opus", - "claude-3-7-sonnet-r" + "claude-opus-4" ] }, "creative-writing": { "overall": 81, "notes": "Competent but reasoning focus may reduce creative spontaneity. Higher latency for creative tasks.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] } }, @@ -643,8 +638,8 @@ "pricing": { "input": "$15.00 per 1M tokens", "output": "$60.00 per 1M tokens", - "notes": "Premium reasoning model pricing, significantly higher than standard models", - "last_verified": "2025-11-09" + "notes": "Premium reasoning model pricing, significantly higher than standard models. Pricing applies until API shutdown 2026-10-23.", + "last_verified": "2026-07-09" }, "context_window": 128000, "languages": [ diff --git a/data/models/openai-o3-mini.json b/data/models/openai-o3-mini.json index 32438ca..65d6298 100644 --- a/data/models/openai-o3-mini.json +++ b/data/models/openai-o3-mini.json @@ -4,7 +4,7 @@ "name": "OpenAI o3-mini", "provider": "OpenAI", "version": "20251201", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", "description": "DEPRECATED: o3-mini's API shuts down 2026-10-23; migration target is GPT-5.5. Historically an efficient reasoning model from OpenAI achieving 50% on SWE-bench and 87.3% on HumanEval, optimized for fast reasoning at competitive pricing with strong coding capabilities.", "website": "https://openai.com/o3-mini/", @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 87, @@ -44,7 +44,7 @@ } ], "methodology": "Competition-level reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 86, @@ -58,7 +58,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 88, @@ -72,7 +72,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.8s", @@ -86,7 +86,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.2s", @@ -100,7 +100,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -114,7 +114,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -128,7 +128,7 @@ } ], "methodology": "Historical data", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong performance with efficient reasoning. Excellent HumanEval at 87.3% with fast latency." @@ -148,7 +148,7 @@ } ], "methodology": "OWASP LLM01 testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -162,7 +162,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -176,7 +176,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 89, @@ -190,7 +190,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 87, @@ -204,7 +204,7 @@ } ], "methodology": "Security review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good security with reasoning-enhanced safety." @@ -224,7 +224,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -238,7 +238,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -252,7 +252,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 80, @@ -266,7 +266,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 86, @@ -280,7 +280,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 80, @@ -294,7 +294,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good privacy with SOC 2. 30-day retention minimum." @@ -314,7 +314,7 @@ } ], "methodology": "Feature evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 85, @@ -328,7 +328,7 @@ } ], "methodology": "QA testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 82, @@ -342,7 +342,7 @@ } ], "methodology": "Bias testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -356,7 +356,7 @@ } ], "methodology": "Confidence assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 89, @@ -370,7 +370,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -384,7 +384,7 @@ } ], "methodology": "Disclosure review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -398,7 +398,7 @@ } ], "methodology": "Safety analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with visible reasoning. Strong safety guardrails." @@ -418,7 +418,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -432,7 +432,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 76, @@ -447,12 +447,12 @@ { "source": "OpenAI Deprecations", "url": "https://developers.openai.com/api/docs/deprecations", - "date": "2026-06-10", - "value": "o3-mini API shutdown 2026-10-23; migration target GPT-5.5" + "date": "2026-07-09", + "value": "o3-mini-2025-01-31 / o3-mini shutdown 2026-10-23; recommended replacement gpt-5.5" } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -466,7 +466,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -480,7 +480,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 83, @@ -494,7 +494,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 89, @@ -508,7 +508,7 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Deprecated: o3-mini API shutdown scheduled 2026-10-23, migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." @@ -520,7 +520,7 @@ "notes": "Strong coding with 87.3% HumanEval. Fast latency great for development workflows.", "alternatives": [ "openai-o4-mini", - "claude-4-sonnet" + "claude-sonnet-4" ] }, "customer-support": { @@ -534,8 +534,7 @@ "overall": 82, "notes": "Adequate but reasoning may be unnecessary for creative tasks.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "data-analysis": { @@ -543,7 +542,7 @@ "notes": "Strong analytical capabilities with efficient reasoning.", "alternatives": [ "openai-o1", - "claude-4-opus" + "claude-opus-4" ] }, "research-assistant": { @@ -551,21 +550,21 @@ "notes": "Good research with visible reasoning at affordable pricing.", "alternatives": [ "openai-o1", - "claude-4-opus" + "claude-opus-4" ] }, "legal-compliance": { "overall": 82, "notes": "Good reasoning but 30-day retention may be concern.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "healthcare": { "overall": 81, "notes": "Not HIPAA eligible by default.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "gemini-2-0-flash" ] }, @@ -574,7 +573,7 @@ "notes": "Strong analytical capabilities at reasonable pricing.", "alternatives": [ "openai-o1", - "claude-4-opus" + "claude-opus-4" ] }, "education": { @@ -582,15 +581,14 @@ "notes": "Excellent for education with visible reasoning and good value.", "alternatives": [ "openai-o4-mini", - "claude-4-opus" + "claude-opus-4" ] }, "creative-writing": { "overall": 79, "notes": "Adequate but reasoning may hinder creativity.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] } }, @@ -628,10 +626,10 @@ ], "metadata": { "pricing": { - "input": "$1.00 per 1M tokens", - "output": "$4.00 per 1M tokens", - "notes": "Budget-friendly reasoning model pricing (Flex tier)", - "last_verified": "2025-11-09" + "input": "$1.10 per 1M tokens", + "output": "$4.40 per 1M tokens", + "notes": "Budget-friendly reasoning model pricing (standard tier; reasoning tokens billed as output). Pricing applies until API shutdown 2026-10-23.", + "last_verified": "2026-07-09" }, "context_window": 128000, "languages": [ diff --git a/data/models/openai-o3.json b/data/models/openai-o3.json index 4ab3972..b8bb34d 100644 --- a/data/models/openai-o3.json +++ b/data/models/openai-o3.json @@ -4,9 +4,9 @@ "name": "OpenAI o3", "provider": "OpenAI", "version": "2025-01", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DEPRECATED: the o3 family is being retired — o3-deep-research shuts down 2026-07-23 and o3-mini's API shuts down 2026-10-23; migration target is GPT-5.5. Historically OpenAI's most advanced reasoning model of its era, with exceptional performance on complex coding and mathematical tasks.", + "description": "DEPRECATED: the entire o3 family now has shutdown dates — o3 (o3-2025-04-16) and o3-pro shut down in the API on 2026-12-11 (announced 2026-06-11), o3-deep-research shuts down 2026-07-23, and o3-mini shuts down 2026-10-23. Migration targets are GPT-5.5 (o3, o3-mini) and GPT-5.5-pro (o3-pro, o3-deep-research). Historically OpenAI's most advanced reasoning model of its era, with exceptional performance on complex coding and mathematical tasks.", "website": "https://openai.com/o3", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks measuring real-world programming tasks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 96, @@ -50,7 +50,7 @@ } ], "methodology": "Advanced reasoning benchmarks requiring multi-step problem solving", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 94, @@ -70,7 +70,7 @@ } ], "methodology": "Crowdsourced blind comparisons and comprehensive knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 93, @@ -84,7 +84,7 @@ } ], "methodology": "Internal testing with repeated prompts at various temperature settings", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Chain-of-thought reasoning provides consistent problem-solving approaches" }, "latency_p50": { @@ -99,7 +99,7 @@ } ], "methodology": "Median latency for API requests with standard prompt sizes", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "6.5s", @@ -113,7 +113,7 @@ } ], "methodology": "95th percentile response time across diverse workloads", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -127,7 +127,7 @@ } ], "methodology": "Official specification from provider", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 98, @@ -141,7 +141,7 @@ } ], "methodology": "Historical uptime data from official status page", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Industry-leading performance on coding and reasoning tasks. Significantly higher latency due to chain-of-thought reasoning process, but delivers exceptional accuracy." @@ -167,7 +167,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection attacks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 89, @@ -187,7 +187,7 @@ } ], "methodology": "Testing against adversarial prompt datasets", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -201,7 +201,7 @@ } ], "methodology": "Analysis of privacy policies and data handling practices", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Strong policies, but inherent LLM memorization risks exist" }, "output_safety": { @@ -216,7 +216,7 @@ } ], "methodology": "Comprehensive safety testing across harmful content categories", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 85, @@ -230,7 +230,7 @@ } ], "methodology": "Review of API security features and best practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong security posture with reasoning-enhanced safety checks. Robust resistance to adversarial attacks." @@ -250,7 +250,7 @@ } ], "methodology": "Review of enterprise documentation and privacy policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -264,7 +264,7 @@ } ], "methodology": "Analysis of privacy policy and data usage terms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -278,7 +278,7 @@ } ], "methodology": "Review of terms of service and data retention policies", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 82, @@ -292,7 +292,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "No automatic PII detection, customers must implement their own controls" }, "compliance_certifications": { @@ -307,7 +307,7 @@ } ], "methodology": "Verification of compliance certifications and audit reports", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 75, @@ -321,7 +321,7 @@ } ], "methodology": "Review of data handling practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good privacy practices with opt-out for training data. 30-day data retention for abuse monitoring is longer than some competitors." @@ -341,7 +341,7 @@ } ], "methodology": "Evaluation of reasoning transparency and explanation capabilities", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 88, @@ -361,7 +361,7 @@ } ], "methodology": "Testing on factual QA datasets and real-world usage", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Chain-of-thought reasoning significantly reduces hallucinations" }, "bias_fairness": { @@ -382,7 +382,7 @@ } ], "methodology": "Evaluation on bias benchmarks and diverse demographic testing", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Ongoing work, but biases still present in some outputs" }, "uncertainty_quantification": { @@ -397,7 +397,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Reasoning process provides natural uncertainty quantification" }, "model_card_quality": { @@ -412,7 +412,7 @@ } ], "methodology": "Review of documentation completeness and clarity", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 74, @@ -426,7 +426,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Limited transparency on specific training data sources (industry standard)" }, "guardrails": { @@ -441,7 +441,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Excellent explainability through chain-of-thought reasoning. Strong hallucination resistance. Training data transparency could be improved." @@ -461,7 +461,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 93, @@ -475,7 +475,7 @@ } ], "methodology": "Review of SDK quality, documentation, and maintenance", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 73, @@ -490,12 +490,12 @@ { "source": "OpenAI Deprecations", "url": "https://developers.openai.com/api/docs/deprecations", - "date": "2026-06-10", - "value": "o3-deep-research shutdown 2026-07-23; o3-mini API shutdown 2026-10-23; migration target GPT-5.5" + "date": "2026-07-09", + "value": "o3-2025-04-16 and o3-pro-2025-06-10 shutdown 2026-12-11 (announced 2026-06-11), replacements gpt-5.5 / gpt-5.5-pro; o3-deep-research shutdown 2026-07-23 (replacement gpt-5.5-pro); o3-mini shutdown 2026-10-23 (replacement gpt-5.5)" } ], "methodology": "Review of versioning policy and historical practices", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 84, @@ -509,7 +509,7 @@ } ], "methodology": "Review of available monitoring tools and metrics", - "last_verified": "2025-11-08", + "last_verified": "2026-07-09", "notes": "Basic metrics available, limited detailed tracing" }, "support_quality": { @@ -524,7 +524,7 @@ } ], "methodology": "Assessment of documentation, community, and support responsiveness", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 84, @@ -538,7 +538,7 @@ } ], "methodology": "Analysis of third-party integrations and tools", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -552,10 +552,10 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, - "notes": "Deprecated: o3-deep-research shuts down 2026-07-23 and o3-mini API shuts down 2026-10-23; migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." + "notes": "Deprecated: o3 and o3-pro API shutdown 2026-12-11 (announced 2026-06-11); o3-deep-research shuts down 2026-07-23 and o3-mini 2026-10-23. Migration targets GPT-5.5 / GPT-5.5-pro. Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { @@ -572,7 +572,7 @@ "notes": "Slower response times make it less ideal for real-time support. Better suited for complex troubleshooting requiring deep reasoning.", "alternatives": [ "gpt-4-1", - "claude-3-5-haiku" + "claude-haiku-4-5" ] }, "content-creation": { @@ -649,11 +649,11 @@ "limitations": [ "Higher latency due to reasoning overhead (~3.2s p50, ~6.5s p95)", "30-day data retention longer than some competitors", - "Premium pricing for reasoning capabilities", + "Reasoning tokens billed as output can multiply effective cost despite $2/$8 list pricing", "Not HIPAA eligible", "Limited regional data residency options", "Reasoning overhead unnecessary for simple tasks", - "DEPRECATED: o3-deep-research shuts down 2026-07-23; o3 family API shutdown 2026-10-23 — migrate to GPT-5.5" + "DEPRECATED: o3 and o3-pro API shutdown 2026-12-11; o3-deep-research shuts down 2026-07-23 and o3-mini 2026-10-23 — migrate to GPT-5.5 / GPT-5.5-pro" ], "best_for": [ "Complex coding tasks requiring advanced algorithms", @@ -671,10 +671,10 @@ ], "metadata": { "pricing": { - "input": "$15.00 per 1M tokens", - "output": "$60.00 per 1M tokens", - "notes": "Premium pricing reflecting advanced reasoning capabilities (pricing varies by variant/tier)", - "last_verified": "2025-11-09" + "input": "$2.00 per 1M tokens", + "output": "$8.00 per 1M tokens", + "notes": "o3 standard pricing after OpenAI's June 2025 price cut (o3-pro is $20/$80). Reasoning tokens are billed as output. Pricing applies until API shutdown 2026-12-11.", + "last_verified": "2026-07-09" }, "context_window": 128000, "languages": [ diff --git a/data/models/openai-o4-mini.json b/data/models/openai-o4-mini.json index aa636e6..f40857c 100644 --- a/data/models/openai-o4-mini.json +++ b/data/models/openai-o4-mini.json @@ -4,9 +4,9 @@ "name": "OpenAI o4-mini", "provider": "OpenAI", "version": "o4-mini-2025-04-16", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "DEPRECATED: removed from ChatGPT 2026-02-13; o4-mini API shuts down 2026-10-23 (o4-mini-deep-research shut down 2026-07-23); migrate to GPT-5.5. Historically OpenAI's best small reasoning model (April 2025): 93% AIME, 68% SWE-bench, first mini with full tool support + multimodality.", + "description": "DEPRECATED: removed from ChatGPT 2026-02-13; o4-mini API shuts down 2026-10-23 (o4-mini-deep-research shuts down 2026-07-23); OpenAI's recommended replacement is gpt-5.4-mini (gpt-5.5-pro for deep-research). Historically OpenAI's best small reasoning model (April 2025): 93% AIME, 68% SWE-bench, first mini with full tool support + multimodality.", "website": "https://openai.com/index/introducing-o3-and-o4-mini/", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Industry-standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 87, @@ -44,7 +44,7 @@ } ], "methodology": "Competition-level reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 86, @@ -58,7 +58,7 @@ } ], "methodology": "Comprehensive knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 88, @@ -72,7 +72,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.8s", @@ -86,7 +86,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.2s", @@ -100,7 +100,7 @@ } ], "methodology": "95th percentile", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "128,000 tokens", @@ -114,7 +114,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 99, @@ -128,7 +128,7 @@ } ], "methodology": "Historical data", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong performance with efficient reasoning. Excellent HumanEval at 87.3% with fast latency." @@ -148,7 +148,7 @@ } ], "methodology": "OWASP LLM01 testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 88, @@ -162,7 +162,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 83, @@ -176,7 +176,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 89, @@ -190,7 +190,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 87, @@ -204,7 +204,7 @@ } ], "methodology": "Security review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good security with reasoning-enhanced safety." @@ -224,7 +224,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 90, @@ -238,7 +238,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "30 days", @@ -252,7 +252,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 80, @@ -266,7 +266,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 86, @@ -280,7 +280,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 80, @@ -294,7 +294,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good privacy with SOC 2. 30-day retention minimum." @@ -314,7 +314,7 @@ } ], "methodology": "Feature evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 85, @@ -328,7 +328,7 @@ } ], "methodology": "QA testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 82, @@ -342,7 +342,7 @@ } ], "methodology": "Bias testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 86, @@ -356,7 +356,7 @@ } ], "methodology": "Confidence assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 89, @@ -370,7 +370,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 78, @@ -384,7 +384,7 @@ } ], "methodology": "Disclosure review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 90, @@ -398,7 +398,7 @@ } ], "methodology": "Safety analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good transparency with visible reasoning. Strong safety guardrails." @@ -418,7 +418,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 92, @@ -432,7 +432,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 76, @@ -447,12 +447,12 @@ { "source": "OpenAI Deprecations", "url": "https://developers.openai.com/api/docs/deprecations", - "date": "2026-06-10", - "value": "o4-mini removed from ChatGPT 2026-02-13; o4-mini-deep-research shutdown 2026-07-23; o4-mini API shutdown 2026-10-23; migration target GPT-5.5" + "date": "2026-07-09", + "value": "o4-mini-2025-04-16 / o4-mini (and ft-o4-mini) API shutdown 2026-10-23, recommended replacement gpt-5.4-mini; o4-mini-deep-research shutdown 2026-07-23, replacement gpt-5.5-pro" } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 87, @@ -466,7 +466,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 88, @@ -480,7 +480,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 83, @@ -494,7 +494,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 89, @@ -508,10 +508,10 @@ } ], "methodology": "Terms review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, - "notes": "Deprecated: removed from ChatGPT 2026-02-13; o4-mini API shutdown scheduled 2026-10-23, migration target GPT-5.5. Versioning and ecosystem scores reduced to reflect deprecation." + "notes": "Deprecated: removed from ChatGPT 2026-02-13; o4-mini API shutdown scheduled 2026-10-23, recommended replacement gpt-5.4-mini (verified against OpenAI deprecations page 2026-07-09). Versioning and ecosystem scores reduced to reflect deprecation." } }, "use_case_ratings": { @@ -520,7 +520,7 @@ "notes": "Strong coding with 87.3% HumanEval. Fast latency great for development workflows.", "alternatives": [ "openai-o4-mini", - "claude-4-sonnet" + "claude-sonnet-4" ] }, "customer-support": { @@ -534,8 +534,7 @@ "overall": 82, "notes": "Adequate but reasoning may be unnecessary for creative tasks.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "data-analysis": { @@ -543,7 +542,7 @@ "notes": "Strong analytical capabilities with efficient reasoning.", "alternatives": [ "openai-o1", - "claude-4-opus" + "claude-opus-4" ] }, "research-assistant": { @@ -551,21 +550,21 @@ "notes": "Good research with visible reasoning at affordable pricing.", "alternatives": [ "openai-o1", - "claude-4-opus" + "claude-opus-4" ] }, "legal-compliance": { "overall": 82, "notes": "Good reasoning but 30-day retention may be concern.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "healthcare": { "overall": 81, "notes": "Not HIPAA eligible by default.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "gemini-2-0-flash" ] }, @@ -574,7 +573,7 @@ "notes": "Strong analytical capabilities at reasonable pricing.", "alternatives": [ "openai-o1", - "claude-4-opus" + "claude-opus-4" ] }, "education": { @@ -582,15 +581,14 @@ "notes": "Excellent for education with visible reasoning and good value.", "alternatives": [ "openai-o4-mini", - "claude-4-opus" + "claude-opus-4" ] }, "creative-writing": { "overall": 79, "notes": "Adequate but reasoning may hinder creativity.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] } }, @@ -609,7 +607,7 @@ "Mini model limitations for complex reasoning", "Reasoning overhead for simple tasks", "Moderate general knowledge (75.8% MMLU)", - "DEPRECATED: removed from ChatGPT 2026-02-13; o4-mini API shutdown 2026-10-23 — migrate to GPT-5.5" + "DEPRECATED: removed from ChatGPT 2026-02-13; o4-mini API shutdown 2026-10-23 — migrate to gpt-5.4-mini" ], "best_for": [ "Code generation on a budget", @@ -628,10 +626,10 @@ ], "metadata": { "pricing": { - "input": "$1.00 per 1M tokens", - "output": "$4.00 per 1M tokens", - "notes": "Budget-friendly reasoning model pricing (Flex tier)", - "last_verified": "2025-11-09" + "input": "$1.10 per 1M tokens", + "output": "$4.40 per 1M tokens", + "notes": "Budget-friendly reasoning model pricing (standard tier; Flex processing is discounted). Pricing applies until API shutdown 2026-10-23.", + "last_verified": "2026-07-09" }, "context_window": 128000, "languages": [ diff --git a/data/models/qwen2-5-vl-32b.json b/data/models/qwen2-5-vl-32b.json index a6a9ae8..8cf4bd3 100644 --- a/data/models/qwen2-5-vl-32b.json +++ b/data/models/qwen2-5-vl-32b.json @@ -4,9 +4,9 @@ "name": "Qwen2.5-VL-32B", "provider": "Alibaba", "version": "20251020", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Multimodal vision-language model from Alibaba, now two generations behind: superseded first by Qwen3-VL (Sep 2025) and then by the natively-multimodal Qwen3.5 (released 2026-02-16). Historically achieved 42.9% on SWE-bench with strong image understanding at competitive pricing; new deployments should evaluate Qwen3.5 instead.", + "description": "Multimodal vision-language model from Alibaba, now three generations behind: superseded by Qwen3-VL (Sep 2025), the natively-multimodal Qwen3.5 (released 2026-02-16), and the multimodal-input Qwen3.6 open models (Apr 2026). Historically achieved 42.9% on SWE-bench with strong image understanding at competitive pricing; new deployments should evaluate Qwen3.5/Qwen3.6 instead.", "website": "https://qwenlm.github.io/", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Standard coding benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 84, @@ -44,7 +44,7 @@ } ], "methodology": "Reasoning benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 88, @@ -58,7 +58,7 @@ } ], "methodology": "Knowledge testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "vision_accuracy": { "score": 92, @@ -72,7 +72,7 @@ } ], "methodology": "Vision-specific benchmarks", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 84, @@ -86,7 +86,7 @@ } ], "methodology": "Internal testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.6s", @@ -100,7 +100,7 @@ } ], "methodology": "Median latency", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "context_window": { "value": "32,768 tokens", @@ -114,7 +114,7 @@ } ], "methodology": "Official specification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uptime": { "score": 96, @@ -128,7 +128,7 @@ } ], "methodology": "Historical data", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Strong vision capabilities with good coding performance. 32B parameter size provides good balance." @@ -148,7 +148,7 @@ } ], "methodology": "OWASP testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 81, @@ -162,7 +162,7 @@ } ], "methodology": "Adversarial testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 78, @@ -176,7 +176,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "output_safety": { "score": 83, @@ -190,7 +190,7 @@ } ], "methodology": "Safety testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "api_security": { "score": 81, @@ -204,7 +204,7 @@ } ], "methodology": "Security review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Adequate security with standard guardrails." @@ -224,7 +224,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 85, @@ -238,7 +238,7 @@ } ], "methodology": "Policy analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "data_retention": { "value": "90 days", @@ -252,7 +252,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 76, @@ -266,7 +266,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 80, @@ -280,7 +280,7 @@ } ], "methodology": "Certification verification", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 73, @@ -294,7 +294,7 @@ } ], "methodology": "Policy review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Limited privacy for Western markets. Asian data residency." @@ -314,7 +314,7 @@ } ], "methodology": "Feature evaluation", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 81, @@ -328,7 +328,7 @@ } ], "methodology": "QA testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 79, @@ -342,7 +342,7 @@ } ], "methodology": "Bias testing", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 80, @@ -356,7 +356,7 @@ } ], "methodology": "Confidence assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 86, @@ -370,7 +370,7 @@ } ], "methodology": "Documentation review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 75, @@ -384,7 +384,7 @@ } ], "methodology": "Disclosure review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "guardrails": { "score": 84, @@ -398,7 +398,7 @@ } ], "methodology": "Safety analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Moderate transparency with standard safety features." @@ -418,7 +418,7 @@ } ], "methodology": "API review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 84, @@ -432,7 +432,7 @@ } ], "methodology": "SDK review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 74, @@ -449,11 +449,17 @@ "url": "https://qwen.ai/blog?id=qwen3.5", "date": "2026-06-10", "value": "Two generations behind: superseded by Qwen3-VL (Sep 2025) and natively-multimodal Qwen3.5 (released 2026-02-16)" + }, + { + "source": "MarkTechPost - Qwen3.6-27B release", + "url": "https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/", + "date": "2026-07-09", + "value": "Now three generations behind: Qwen3.6 open models (35B-A3B on 2026-04-16, 27B dense on 2026-04-22, Apache 2.0, 256K context, text/image/video input) further supersede this line" } ], "methodology": "Policy review", - "last_verified": "2026-06-10", - "notes": "Open weights remain available, but the line has moved on to Qwen3.5" + "last_verified": "2026-07-09", + "notes": "Open weights remain available, but the line has moved on to Qwen3.5/Qwen3.6" }, "monitoring_observability": { "score": 81, @@ -467,7 +473,7 @@ } ], "methodology": "Tool review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "support_quality": { "score": 82, @@ -481,7 +487,7 @@ } ], "methodology": "Support assessment", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 80, @@ -495,7 +501,7 @@ } ], "methodology": "Ecosystem analysis", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" }, "license_terms": { "score": 90, @@ -509,7 +515,7 @@ } ], "methodology": "License review", - "last_verified": "2025-11-08" + "last_verified": "2026-07-09" } }, "notes": "Good operational quality with open-source license." @@ -520,7 +526,7 @@ "overall": 86, "notes": "Good coding with vision support. Useful for UI/UX code generation from images.", "alternatives": [ - "claude-4-sonnet", + "claude-sonnet-4", "openai-o4-mini" ] }, @@ -535,8 +541,7 @@ "overall": 84, "notes": "Good for content with visual elements. Can describe and analyze images.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] }, "data-analysis": { @@ -550,21 +555,21 @@ "overall": 87, "notes": "Strong for research with visual materials. Can analyze papers with diagrams.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "legal-compliance": { "overall": 75, "notes": "Limited compliance for Western markets. Data residency concerns.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "healthcare": { "overall": 82, "notes": "Good for medical image analysis but limited Western compliance.", "alternatives": [ - "claude-4-opus", + "claude-opus-4", "gemini-2-0-flash" ] }, @@ -572,7 +577,7 @@ "overall": 83, "notes": "Good for analyzing financial charts and visual reports.", "alternatives": [ - "claude-4-opus" + "claude-opus-4" ] }, "education": { @@ -586,8 +591,7 @@ "overall": 82, "notes": "Good for visual storytelling and image-based creative content.", "alternatives": [ - "claude-4-opus", - "gpt-4-5" + "claude-opus-4" ] } }, @@ -606,7 +610,7 @@ "Smaller context window (32K tokens)", "Lower coding benchmarks (42.9% SWE-bench)", "Growing but less mature ecosystem", - "Superseded: two generations behind, replaced by Qwen3-VL (Sep 2025) and the natively-multimodal Qwen3.5 (released 2026-02-16)" + "Superseded: three generations behind, replaced by Qwen3-VL (Sep 2025), the natively-multimodal Qwen3.5 (2026-02-16), and the Qwen3.6 open models (Apr 2026)" ], "best_for": [ "Visual AI applications requiring image understanding", @@ -627,7 +631,7 @@ "pricing": { "input": "$0.40 per 1M tokens", "output": "$1.20 per 1M tokens", - "notes": "Highly competitive pricing for vision model" + "notes": "Historical DashScope pricing for a superseded model; current hosted availability and rates not re-verified — self-hosting from Apache-2.0 weights remains the reliable route" }, "context_window": 32768, "languages": [ diff --git a/data/models/qwen3-5.json b/data/models/qwen3-5.json index f9a5632..8ce7d55 100644 --- a/data/models/qwen3-5.json +++ b/data/models/qwen3-5.json @@ -4,9 +4,9 @@ "name": "Qwen3.5", "provider": "Alibaba", "version": "20260216", - "last_evaluated": "2026-06-10", + "last_evaluated": "2026-07-09", "evaluated_by": "TrustVector Team", - "description": "Alibaba's Apache-2.0 flagship open model: Qwen3.5-397B-A17B, a hybrid MoE with 512 experts (397B total / 17B active) that is natively multimodal, supports 262K context (1M on hosted Qwen3.5-Plus) and 201 languages, and beats Alibaba's own API-only 1T-parameter Qwen3-Max while decoding up to 19x faster at long context.", + "description": "Alibaba's Apache-2.0 flagship open model: Qwen3.5-397B-A17B, a 512-expert hybrid MoE (397B total / 17B active), natively multimodal, 262K context (1M on hosted Qwen3.5-Plus), 201 languages; beats Alibaba's API-only 1T Qwen3-Max with up to 19x faster long-context decode. Still its largest open-weight model as of July 2026, but smaller Apache-2.0 Qwen3.6 models (Apr 2026) surpass it on agentic coding, and the newest frontier (Qwen3.7-Max, May 2026; Qwen3.7-Plus, Jun 2026) is API-only.", "website": "https://qwen.ai/blog?id=qwen3.5", "trust_vector": { "performance_reliability": { @@ -30,7 +30,7 @@ } ], "methodology": "Vendor benchmarks corroborated by independent press coverage and community leaderboards", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_reasoning": { "score": 93, @@ -50,7 +50,7 @@ } ], "methodology": "Mathematical and agentic reasoning benchmarks from the model card and release blog, cross-checked against community evaluations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "task_accuracy_general": { "score": 92, @@ -64,7 +64,7 @@ } ], "methodology": "Comprehensive knowledge and multimodal benchmark review including multilingual coverage", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_consistency": { "score": 88, @@ -78,7 +78,7 @@ } ], "methodology": "Repeated-prompt testing across temperature settings, supplemented by community reports", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p50": { "value": "1.5s (hosted); 17B active params enable fast self-hosted decode", @@ -92,7 +92,7 @@ } ], "methodology": "Median latency on hosted endpoints and decode-throughput comparisons from independent reporting", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "latency_p95": { "value": "3.8s (hosted)", @@ -106,7 +106,7 @@ } ], "methodology": "95th percentile response time across diverse workloads from independent benchmarking", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": { "value": "262,144 tokens (1M on hosted Qwen3.5-Plus)", @@ -120,7 +120,7 @@ } ], "methodology": "Official specification from model card and Alibaba Cloud documentation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uptime": { "score": 95, @@ -134,7 +134,7 @@ } ], "methodology": "Hosted-platform availability history plus redundancy across third-party hosts", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Beats Alibaba's own 1T-parameter API-only Qwen3-Max with only 17B active parameters, with up to 19x faster decode at 256K context. Native multimodality and 201-language coverage are unmatched among open models." @@ -154,7 +154,7 @@ } ], "methodology": "Testing against OWASP LLM01 prompt injection patterns, including image-borne injection for multimodal inputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "jailbreak_resistance": { "score": 84, @@ -168,7 +168,7 @@ } ], "methodology": "Adversarial prompt testing; assessment accounts for open-weight modifiability", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_leakage_prevention": { "score": 81, @@ -182,7 +182,7 @@ } ], "methodology": "Analysis of hosted-platform policies plus the self-hosting option for full data isolation", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "output_safety": { "score": 86, @@ -196,7 +196,7 @@ } ], "methodology": "Safety testing across harmful content categories and multiple languages on default weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "api_security": { "score": 86, @@ -210,7 +210,7 @@ } ], "methodology": "Review of API security features on the first-party hosted platform", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Solid default guardrails with notably broad multilingual safety coverage. Multimodal inputs widen the attack surface; open weights shift responsibility to deployers who fine-tune." @@ -230,7 +230,7 @@ } ], "methodology": "Review of hosting regions and licensing; China-jurisdiction caveat applies to Alibaba's first-party API, not self-hosted or Western-hosted deployments", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_optout": { "score": 86, @@ -244,7 +244,7 @@ } ], "methodology": "Analysis of hosted-platform data usage terms", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "data_retention": { "value": "Per Alibaba Cloud policy (region-dependent); zero when self-hosted", @@ -258,7 +258,7 @@ } ], "methodology": "Review of hosted-platform retention policies; retention is deployment-dependent for open-weight models", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "pii_handling": { "score": 77, @@ -272,7 +272,7 @@ } ], "methodology": "Review of data protection capabilities and customer responsibilities", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "compliance_certifications": { "score": 74, @@ -286,7 +286,7 @@ } ], "methodology": "Verification of infrastructure certifications versus model-service-level compliance for Western regulated markets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "zero_data_retention": { "score": 78, @@ -300,7 +300,7 @@ } ], "methodology": "Review of data handling across first-party API, third-party hosts, and self-hosting", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Alibaba's first-party API is China-jurisdiction (Singapore region available), which concerns Western regulated buyers; Apache-2.0 self-hosting or Western third-party hosting fully avoids that. Alibaba Cloud's infrastructure certifications are stronger than DeepSeek's platform but still lack HIPAA/FedRAMP for the model service." @@ -320,7 +320,7 @@ } ], "methodology": "Evaluation of reasoning transparency and trace accessibility", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "hallucination_rate": { "score": 82, @@ -334,7 +334,7 @@ } ], "methodology": "Testing on factual QA and multimodal grounding datasets", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "bias_fairness": { "score": 78, @@ -348,7 +348,7 @@ } ], "methodology": "Evaluation on bias benchmarks across languages and politically sensitive topic probes", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "uncertainty_quantification": { "score": 78, @@ -362,7 +362,7 @@ } ], "methodology": "Qualitative assessment of confidence expression in outputs", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "model_card_quality": { "score": 90, @@ -376,7 +376,7 @@ } ], "methodology": "Review of model card and technical documentation completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "training_data_transparency": { "score": 76, @@ -390,7 +390,7 @@ } ], "methodology": "Review of public disclosures about training data", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "guardrails": { "score": 82, @@ -404,7 +404,7 @@ } ], "methodology": "Analysis of built-in safety mechanisms in default weights", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, "notes": "Strong open documentation and inspectable reasoning. Typical open-model gaps remain: limited training-data detail and topic-avoidance on politically sensitive subjects in default weights." @@ -424,7 +424,7 @@ } ], "methodology": "Review of API design, consistency, and feature completeness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "sdk_quality": { "score": 87, @@ -438,7 +438,7 @@ } ], "methodology": "Review of SDK and inference-framework support", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "versioning_policy": { "score": 82, @@ -449,10 +449,22 @@ "url": "https://qwen.ai/blog?id=qwen3.5", "date": "2026-02-16", "value": "Fast release cadence (supersedes Qwen3 family and Qwen2.5-VL); open weights remain permanently available, softening deprecation impact" + }, + { + "source": "MarkTechPost - Qwen3.6-27B release", + "url": "https://www.marktechpost.com/2026/04/22/alibaba-qwen-team-releases-qwen3-6-27b-a-dense-open-weight-model-outperforming-397b-moe-on-agentic-coding-benchmarks/", + "date": "2026-04-22", + "value": "Qwen3.6 open models shipped Apr 2026 (35B-A3B on 04-16, 27B dense on 04-22, Apache 2.0, 256K context): the 27B dense outperforms the 397B-A17B Qwen3.5 flagship on agentic coding benchmarks" + }, + { + "source": "innFactory / EQS News - Qwen3.7 releases", + "url": "https://innfactory.ai/en/ai-models/qwen/", + "date": "2026-07-09", + "value": "Family frontier moved to proprietary API-only models: Qwen3.7-Max (2026-05-20, Apsara Summit) and Qwen3.7-Plus (2026-06-01), 1M context, agentic positioning; Alibaba keeps its newest flagship closed-weight" } ], "methodology": "Review of release cadence and weight-availability guarantees", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "monitoring_observability": { "score": 84, @@ -466,7 +478,7 @@ } ], "methodology": "Review of monitoring tools across deployment options", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "support_quality": { "score": 78, @@ -480,7 +492,7 @@ } ], "methodology": "Assessment of support tiers, documentation, and community responsiveness", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "ecosystem_maturity": { "score": 92, @@ -494,7 +506,7 @@ } ], "methodology": "Analysis of derivative models, third-party hosting, and tooling integrations", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "license_terms": { "score": 98, @@ -508,10 +520,10 @@ } ], "methodology": "Review of licensing terms and restrictions", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" } }, - "notes": "Best-in-class open-model ecosystem: Apache 2.0 with patent grant, day-one inference-framework support, and a complete size ladder (0.8B to 397B-A17B) for matching capability to hardware. Supersedes the Qwen3 family and Qwen2.5-VL." + "notes": "Best-in-class open-model ecosystem: Apache 2.0 with patent grant, day-one inference-framework support, and a complete size ladder (0.8B to 397B-A17B) for matching capability to hardware. Supersedes the Qwen3 family and Qwen2.5-VL. Since April-June 2026 the family has continued with the open Qwen3.6 models (Apache 2.0) and the proprietary API-only Qwen3.7-Max/Plus." } }, "use_case_ratings": { @@ -577,6 +589,7 @@ ], "limitations": [ "First-party Alibaba Cloud hosting is China-jurisdiction (Singapore region available); no HIPAA/FedRAMP path for the model service — self-hosting or Western hosts avoid this", + "No longer the family's coding leader: the much smaller Apache-2.0 Qwen3.6-27B (Apr 2026) outperforms it on agentic coding benchmarks, and the newest family frontier (Qwen3.7-Max/Plus) is API-only", "1M context requires the hosted Qwen3.5-Plus; open weights cap at 262K", "Topic-avoidance on politically sensitive subjects in default weights", "Training-data composition disclosed only at a high level", @@ -601,7 +614,7 @@ "input": "Free weights (Apache 2.0); hosted from ~$0.40 per 1M tokens on Alibaba Cloud Model Studio", "output": "Hosted from ~$1.20 per 1M tokens; third-party hosts vary", "notes": "Self-hosting is infrastructure-cost-only; the 17B-active design keeps serving costs low for its capability class. Hosted Qwen3.5-Plus (1M context) priced separately.", - "last_verified": "2026-06-10" + "last_verified": "2026-07-09" }, "context_window": 262144, "max_output": 65536, From ca3604fdb617ed781b1723651da9a6978e99b445 Mon Sep 17 00:00:00 2001 From: JBAhire Date: Thu, 9 Jul 2026 22:52:56 -0700 Subject: [PATCH 2/7] =?UTF-8?q?feat:=2040=20new=20evaluations=20=E2=80=94?= =?UTF-8?q?=208=20models,=2017=20agents,=2015=20MCP=20servers?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Models: Claude Sonnet 5, GPT-5.6 (Sol/Terra/Luna), Grok 4.5, GLM-5.2, MiniMax M3, Kimi K2.7-Code, Nemotron 3 Ultra, Qwen3.6. Agents: Claude Cowork, OpenClaw, ChatGPT Agent, Microsoft Scout, Perplexity Comet, Poke, Trae, Cline, OpenCode, Goose, Warp, Junie, Factory Droids, Replit Agent, Lovable, Antigravity, Kiro. MCPs: Snowflake, Databricks, Salesforce, HubSpot, Asana, Box, Shopify, PayPal, Canva, Grafana, Neon, Exa, Browserbase, ClickHouse, Neo4j. Registry grows 156 -> 196. All scores evidence-linked and calibrated against existing siblings. Co-Authored-By: Claude Fable 5 --- data/agents/amazon-kiro.json | 490 ++++++++++++++++++ data/agents/bytedance-trae.json | 482 ++++++++++++++++++ data/agents/chatgpt-agent.json | 477 ++++++++++++++++++ data/agents/claude-cowork.json | 480 ++++++++++++++++++ data/agents/cline.json | 479 ++++++++++++++++++ data/agents/factory-droids.json | 474 ++++++++++++++++++ data/agents/google-antigravity.json | 481 ++++++++++++++++++ data/agents/goose.json | 474 ++++++++++++++++++ data/agents/jetbrains-junie.json | 467 ++++++++++++++++++ data/agents/lovable.json | 471 ++++++++++++++++++ data/agents/microsoft-scout.json | 490 ++++++++++++++++++ data/agents/openclaw.json | 524 ++++++++++++++++++++ data/agents/opencode.json | 484 ++++++++++++++++++ data/agents/perplexity-comet.json | 472 ++++++++++++++++++ data/agents/poke.json | 466 ++++++++++++++++++ data/agents/replit-agent.json | 497 +++++++++++++++++++ data/agents/warp.json | 481 ++++++++++++++++++ data/mcps/mcp-server-asana.json | 520 ++++++++++++++++++++ data/mcps/mcp-server-box.json | 461 +++++++++++++++++ data/mcps/mcp-server-browserbase.json | 477 ++++++++++++++++++ data/mcps/mcp-server-canva.json | 441 +++++++++++++++++ data/mcps/mcp-server-clickhouse.json | 443 +++++++++++++++++ data/mcps/mcp-server-databricks.json | 441 +++++++++++++++++ data/mcps/mcp-server-exa.json | 444 +++++++++++++++++ data/mcps/mcp-server-grafana.json | 488 ++++++++++++++++++ data/mcps/mcp-server-hubspot.json | 472 ++++++++++++++++++ data/mcps/mcp-server-neo4j.json | 445 +++++++++++++++++ data/mcps/mcp-server-neon.json | 445 +++++++++++++++++ data/mcps/mcp-server-paypal.json | 447 +++++++++++++++++ data/mcps/mcp-server-salesforce.json | 492 ++++++++++++++++++ data/mcps/mcp-server-shopify.json | 452 +++++++++++++++++ data/mcps/mcp-server-snowflake.json | 445 +++++++++++++++++ data/models/claude-sonnet-5.json | 673 +++++++++++++++++++++++++ data/models/glm-5-2.json | 659 +++++++++++++++++++++++++ data/models/gpt-5-6.json | 661 +++++++++++++++++++++++++ data/models/grok-4-5.json | 661 +++++++++++++++++++++++++ data/models/kimi-k2-7-code.json | 640 ++++++++++++++++++++++++ data/models/minimax-m3.json | 659 +++++++++++++++++++++++++ data/models/nemotron-3-ultra.json | 684 ++++++++++++++++++++++++++ data/models/qwen3-6.json | 652 ++++++++++++++++++++++++ 40 files changed, 20391 insertions(+) create mode 100644 data/agents/amazon-kiro.json create mode 100644 data/agents/bytedance-trae.json create mode 100644 data/agents/chatgpt-agent.json create mode 100644 data/agents/claude-cowork.json create mode 100644 data/agents/cline.json create mode 100644 data/agents/factory-droids.json create mode 100644 data/agents/google-antigravity.json create mode 100644 data/agents/goose.json create mode 100644 data/agents/jetbrains-junie.json create mode 100644 data/agents/lovable.json create mode 100644 data/agents/microsoft-scout.json create mode 100644 data/agents/openclaw.json create mode 100644 data/agents/opencode.json create mode 100644 data/agents/perplexity-comet.json create mode 100644 data/agents/poke.json create mode 100644 data/agents/replit-agent.json create mode 100644 data/agents/warp.json create mode 100644 data/mcps/mcp-server-asana.json create mode 100644 data/mcps/mcp-server-box.json create mode 100644 data/mcps/mcp-server-browserbase.json create mode 100644 data/mcps/mcp-server-canva.json create mode 100644 data/mcps/mcp-server-clickhouse.json create mode 100644 data/mcps/mcp-server-databricks.json create mode 100644 data/mcps/mcp-server-exa.json create mode 100644 data/mcps/mcp-server-grafana.json create mode 100644 data/mcps/mcp-server-hubspot.json create mode 100644 data/mcps/mcp-server-neo4j.json create mode 100644 data/mcps/mcp-server-neon.json create mode 100644 data/mcps/mcp-server-paypal.json create mode 100644 data/mcps/mcp-server-salesforce.json create mode 100644 data/mcps/mcp-server-shopify.json create mode 100644 data/mcps/mcp-server-snowflake.json create mode 100644 data/models/claude-sonnet-5.json create mode 100644 data/models/glm-5-2.json create mode 100644 data/models/gpt-5-6.json create mode 100644 data/models/grok-4-5.json create mode 100644 data/models/kimi-k2-7-code.json create mode 100644 data/models/minimax-m3.json create mode 100644 data/models/nemotron-3-ultra.json create mode 100644 data/models/qwen3-6.json diff --git a/data/agents/amazon-kiro.json b/data/agents/amazon-kiro.json new file mode 100644 index 0000000..a8e7ca7 --- /dev/null +++ b/data/agents/amazon-kiro.json @@ -0,0 +1,490 @@ +{ + "id": "amazon-kiro", + "type": "agent", + "name": "Kiro", + "provider": "Amazon Web Services (AWS)", + "version": "GA (IDE + CLI)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "AWS's spec-driven agentic IDE and CLI: it turns prompts into structured specs (requirements.md in EARS notation, design.md, tasks.md) before implementing, alongside a freeform vibe mode, agent hooks, and steering files. Preview 2025-07-14, GA 2025-11-17; official successor to Amazon Q Developer. Available in AWS GovCloud (US) with IAM Identity Center integration, signaling a regulated-workload posture (FedRAMP High authorization in progress, not yet granted).", + "website": "https://kiro.dev/", + "trust_vector": { + "performance_reliability": { + "overall_score": 75, + "criteria": { + "task_completion_accuracy": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Blog - General Availability", + "url": "https://kiro.dev/blog/general-availability/", + "date": "2025-11-17", + "value": "GA (2025-11-17) added property-based testing that validates implementations against spec correctness, checkpoints, and a terminal CLI; spec grounding measurably reduces off-target implementations versus pure prompting" + } + ], + "methodology": "Assessment of implementation accuracy on spec-driven workflows from vendor documentation and independent reviews; vibe mode performs comparably to peer agents", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Documentation", + "url": "https://kiro.dev/docs/", + "date": "2026-06-01", + "value": "Agent reliably drives file edits, shell execution, MCP tools, and agent hooks (event-triggered automations on file save/create) across IDE and CLI" + } + ], + "methodology": "Review of agent tool loop reliability across editor, terminal, hooks, and MCP integrations", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Documentation - Specs", + "url": "https://kiro.dev/docs/specs/", + "date": "2026-06-01", + "value": "Spec-driven development is the product's core: requirements.md (EARS notation), design.md, and tasks.md decompose work into reviewable, sequenced implementation tasks before any code is written" + } + ], + "methodology": "Evaluation of structured planning artifacts and task sequencing; the strongest explicit-planning model among evaluated coding agents", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Documentation - Steering", + "url": "https://kiro.dev/docs/steering/", + "date": "2026-06-01", + "value": "Steering files persist project conventions and context across sessions; specs themselves are durable artifacts that carry design intent between tasks and contributors" + } + ], + "methodology": "Review of steering files, spec persistence, and cross-session context retention", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Blog - General Availability", + "url": "https://kiro.dev/blog/general-availability/", + "date": "2025-11-17", + "value": "Iterates on failing tests and task-level errors; GA checkpoints allow rolling back agent work. Autopilot mode can pursue unproductive paths, consuming credits until interrupted" + } + ], + "methodology": "Assessment of autonomous iteration, checkpoint rollback, and documented failure modes", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Documentation - CLI", + "url": "https://kiro.dev/docs/cli/", + "date": "2026-06-01", + "value": "Custom agents with scoped permissions, context, and prompts run specialized tasks in the CLI; orchestration is task-parallel rather than a coordinated multi-agent manager" + } + ], + "methodology": "Review of custom agent profiles and parallel execution capabilities", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 63, + "criteria": { + "tool_sandboxing": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "AWS Security Bulletin AWS-2025-019", + "url": "https://aws.amazon.com/security/security-bulletins/AWS-2025-019/", + "date": "2025-10-07", + "value": "Prompt-injection paths could execute code without human-in-the-loop confirmation in Autopilot and Supervised modes; fixed in Kiro 0.1.42 (2025-08-01) by requiring HITL confirmation in Supervised mode" + }, + { + "source": "Kiro Documentation - CLI Security", + "url": "https://kiro.dev/docs/cli/chat/security/", + "date": "2026-06-01", + "value": "Supervised mode requires approval for file changes and commands with a trusted-commands allowlist; Autopilot executes without per-step approval and runs directly on the user's machine (no VM/container isolation)" + } + ], + "methodology": "Security review of the approval model; supervised defaults and allowlists are sound, but local execution without OS-level sandboxing and an opt-in full-autonomy mode limit the score", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Documentation - IAM", + "url": "https://kiro.dev/docs/enterprise/iam/", + "date": "2026-06-01", + "value": "Enterprise sign-in via AWS IAM Identity Center with subscriptions (Pro/Pro+/Power) assigned and managed centrally from the AWS Management Console; individuals use AWS Builder ID or social login" + }, + { + "source": "AWS - Kiro now available in AWS GovCloud (US)", + "url": "https://aws.amazon.com/about-aws/whats-new/2026/02/kiro-launch-aws-govcloud-us/", + "date": "2026-02-16", + "value": "GovCloud (US-East/US-West) availability since 2026-02-16 mandates IAM Identity Center authentication, bringing Kiro inside government compliance boundaries" + } + ], + "methodology": "Review of identity integration and centralized administration; inheriting AWS IAM Identity Center is best-in-class among coding agents", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "Embrace The Red - AWS Kiro: Arbitrary Code Execution via Indirect Prompt Injection", + "url": "https://embracethered.com/blog/posts/2025/aws-kiro-aribtrary-command-execution-with-indirect-prompt-injection/", + "date": "2025-09-15", + "value": "Demonstrated indirect prompt injection to arbitrary command execution (since fixed); AWS-2025-019 formalized the fixes with HITL requirements and invisible-control-character handling" + }, + { + "source": "NVD - CVE-2026-0830", + "url": "https://nvd.nist.gov/vuln/detail/CVE-2026-0830", + "date": "2026-01-20", + "value": "CVE-2026-0830: command injection via crafted workspace folder names in the GitLab helper enabled RCE; fixed in Kiro 0.6.18" + } + ], + "methodology": "Assessment of demonstrated injection paths against AWS's remediation record; multiple real flaws, but consistently patched with published security bulletins and researcher credit", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Documentation - Data Protection", + "url": "https://kiro.dev/docs/privacy-and-security/data-protection/", + "date": "2026-06-01", + "value": "TLS 1.2+ in transit, AWS KMS encryption at rest, AWS shared-responsibility model; GovCloud regions keep traffic and data within the government boundary" + } + ], + "methodology": "Data architecture review of encryption, regional isolation, and AWS infrastructure inheritance", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 40, + "confidence": "high", + "evidence": [ + { + "source": "Kiro FAQ", + "url": "https://kiro.dev/faq/", + "date": "2026-06-01", + "value": "Built on Code OSS with Open VSX extension compatibility, but the agent, orchestration, and service are proprietary; AWS publishes security bulletins (AWS-2025-019) and a public GitHub issue tracker" + } + ], + "methodology": "Source availability assessment crediting the OSS base, public issue tracker, and formal security-bulletin practice", + "last_verified": "2026-07-09" + } + }, + "notes": "Kiro has had real vulnerabilities (AWS-2025-019 prompt injection, CVE-2026-0830 RCE), but AWS's disclosure discipline — numbered security bulletins, prompt fixes, researcher credit — plus GovCloud/IAM controls give it a stronger institutional security posture than most agent-IDE peers." + }, + "privacy_compliance": { + "overall_score": 62, + "criteria": { + "data_retention": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Documentation - Data Protection", + "url": "https://kiro.dev/docs/privacy-and-security/data-protection/", + "date": "2026-06-01", + "value": "Free-tier and social-login users have telemetry AND content collection (prompts, code snippets) enabled by default for service improvement, with opt-out toggles; paid users via IAM Identity Center or external IdP are automatically excluded from content collection" + } + ], + "methodology": "Review of default collection settings by tier; enterprise defaults are strong, consumer opt-out-required defaults cost points", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Blog - Kiro loves regulated workloads", + "url": "https://kiro.dev/blog/introducing-govcloud/", + "date": "2026-02-18", + "value": "GovCloud (US) availability targets FedRAMP High and DoD CC SRG alignment (authorization in progress, not yet granted); Kiro inherits AWS's compliance program breadth (SOC, ISO, GDPR DPA) as an AWS service" + } + ], + "methodology": "Compliance posture assessment; AWS program inheritance and the GovCloud boundary are credited, pending product-specific FedRAMP authorization", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Documentation - Models", + "url": "https://kiro.dev/docs/models/", + "date": "2026-06-01", + "value": "Prompts and code are processed by Anthropic Claude models (Sonnet 4.5/4.6, Haiku 4.5, Opus 4.6+) and the Auto router within AWS-managed infrastructure; Bedrock routing is press-reported rather than officially documented" + } + ], + "methodology": "Data flow analysis of model routing; AWS-managed processing narrows the third-party surface versus multi-vendor routing", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 42, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Documentation", + "url": "https://kiro.dev/docs/", + "date": "2026-06-01", + "value": "IDE and CLI run locally but all model inference is cloud-based; no offline mode. GovCloud regions offer a compliance-boundary alternative rather than self-hosting" + } + ], + "methodology": "Deployment options assessment with partial credit for the GovCloud regional option", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 67, + "criteria": { + "documentation_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Documentation", + "url": "https://kiro.dev/docs/", + "date": "2026-06-01", + "value": "Comprehensive docs across IDE, CLI, web, and autonomous agent surfaces, including dedicated privacy/security and data-protection sections per surface, billing docs, and a public changelog" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Blog - General Availability", + "url": "https://kiro.dev/blog/general-availability/", + "date": "2025-11-17", + "value": "Task-by-task execution against tasks.md with diffs, checkpoints for rollback, and hunk-level approval in Supervised mode make agent work reviewable step by step" + } + ], + "methodology": "Review of task-level visibility, diff review workflow, and checkpoint history", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Documentation - Specs", + "url": "https://kiro.dev/docs/specs/", + "date": "2026-06-01", + "value": "Requirements in EARS notation, an explicit design document, and a sequenced task list mean the agent's intent is written down and human-editable before execution — the most explainable workflow in the category" + } + ], + "methodology": "Assessment of upfront rationale artifacts; spec-driven development is inherently an explainability mechanism", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "Kiro FAQ", + "url": "https://kiro.dev/faq/", + "date": "2026-06-01", + "value": "Proprietary agent and service on a Code OSS base; no published agent internals or model details beyond the models list" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Blog - The wait(list) is over", + "url": "https://kiro.dev/blog/waitlist-is-over/", + "date": "2025-10-01", + "value": "Hundreds of thousands joined the preview waitlist within 90 days of the July 2025 launch; active public GitHub issue tracker and Discord, though community trust took a hit in the August 2025 pricing episode" + } + ], + "methodology": "Community engagement analysis across waitlist demand, issue tracker, and Discord activity", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 72, + "criteria": { + "ease_of_integration": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Documentation", + "url": "https://kiro.dev/docs/", + "date": "2026-06-01", + "value": "VS Code settings and Open VSX extension compatibility ease migration; sign-in supports social login, AWS Builder ID, or IAM Identity Center. Spec-first workflow has a learning curve versus chat-first tools" + } + ], + "methodology": "Onboarding and integration friction assessment", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Blog - General Availability", + "url": "https://kiro.dev/blog/general-availability/", + "date": "2025-11-17", + "value": "Team plans with centralized management, a CLI for automation, and AWS-scale service infrastructure; execution remains local-machine-bound rather than cloud-parallel" + } + ], + "methodology": "Scalability assessment of team administration and AWS service backing", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Pricing", + "url": "https://kiro.dev/pricing/", + "date": "2026-07-07", + "value": "Free 50 credits/mo; Pro $20 (1,000 credits); Pro+ $40 (2,000); Pro Max $100 (5,000); Power $200 (10,000); overage $0.04/credit; model choice changes credit burn (Sonnet 4.6 ~1.3x Auto). GovCloud pricing ~20% higher with no free tier" + }, + { + "source": "The Register - AWS pricing for Kiro dev tool 'a wallet-wrecking tragedy'", + "url": "https://www.theregister.com/2025/08/18/aws_updated_kiro_pricing/", + "date": "2025-08-18", + "value": "August 2025 pricing rollout drew heavy backlash amid a metering bug consuming 4-6x expected requests; AWS waived and refunded August charges and later restructured to the current credit tiers" + } + ], + "methodology": "Pricing model analysis; current tiers are clearer, but variable per-task credit burn, model multipliers, and the 2025 metering episode temper predictability", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Kiro Documentation - IAM", + "url": "https://kiro.dev/docs/enterprise/iam/", + "date": "2026-06-01", + "value": "Admins assign and monitor subscriptions from the AWS Management Console; per-user credit usage dashboards and centralized billing through AWS accounts" + } + ], + "methodology": "Monitoring and usage governance assessment leveraging AWS console integration", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Kiro Blog - General Availability", + "url": "https://kiro.dev/blog/general-availability/", + "date": "2025-11-17", + "value": "GA since 2025-11-17 with AWS backing as the official successor to Amazon Q Developer (Q signups closed 2026-05-15, end-of-support 2027-04-30; newest Claude models Kiro-exclusive from 2026-05-29); GovCloud availability signals long-term enterprise commitment" + } + ], + "methodology": "Product maturity and vendor commitment assessment; GA status, succession positioning, and regulated-market investment all indicate durability", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 80, + "notes": "Spec-driven workflow excels on well-defined features and team codebases where design intent must be documented; vibe mode covers quick tasks" + }, + "data-analysis": { + "overall": 66, + "notes": "Solid for building analysis pipelines with AWS service integration, though not an analytics tool itself" + }, + "education": { + "overall": 72, + "notes": "Specs teach requirements thinking and the free tier lowers the barrier, making it unusually good at showing how professional software gets planned" + }, + "research-assistant": { + "overall": 58, + "notes": "MCP tools and steering help with technical research inside projects, but general research is outside its design focus" + } + }, + "best_for": [ + "Teams that want documented, reviewable design intent (requirements/design/tasks) before agent execution", + "AWS-centric enterprises managing agent access through IAM Identity Center and the AWS console", + "Government agencies and contractors needing an agentic IDE inside AWS GovCloud (US) compliance boundaries", + "Organizations migrating from Amazon Q Developer to its designated successor" + ], + "strengths": [ + "Spec-driven development (EARS requirements, design docs, task lists) is the most explainable and auditable agent workflow in the category", + "Best-in-class enterprise identity: IAM Identity Center integration with centralized AWS console administration", + "AWS GovCloud (US) availability (since 2026-02-16) with FedRAMP High/DoD CC SRG alignment in progress — unique among agentic IDEs", + "Disciplined security response: numbered AWS security bulletins (AWS-2025-019), prompt fixes (0.1.42, 0.6.18), researcher credit", + "Enterprise/IdP users are automatically excluded from telemetry and content collection", + "Strong vendor commitment as the official Amazon Q Developer successor, with GA team plans, CLI, and property-based spec testing" + ], + "limitations": [ + "Real vulnerability history: AWS-2025-019 prompt-injection code execution and CVE-2026-0830 workspace-name RCE (both patched)", + "Autopilot mode executes commands without per-step approval on the local machine, with no OS-level sandbox", + "Free-tier/social-login defaults collect prompts and code content for service improvement unless opted out", + "Credit consumption varies by task complexity and model choice; the August 2025 metering bug and pricing backlash damaged cost trust", + "Cloud-only inference with no self-hosted option; Claude-only model lineup", + "Spec-first workflow adds overhead for small tasks, and date reporting on its GA has been muddied by conflicting secondary sources (GA verified as 2025-11-17; a May 2026 GA date circulating in SEO articles conflates later Q Developer retirement milestones)" + ], + "metadata": { + "license": "Proprietary (built on Code OSS; Open VSX extension ecosystem)", + "supported_models": [ + "Auto (default router)", + "Anthropic Claude Sonnet 4.5/4.6", + "Anthropic Claude Haiku 4.5", + "Anthropic Claude Opus 4.5/4.6/4.7 (newest Claude models Kiro-exclusive within AWS tooling from 2026-05-29)" + ], + "programming_languages": [ + "All languages supported by the VS Code ecosystem" + ], + "deployment_type": "Local IDE (Code OSS fork) + CLI with cloud model inference; commercial AWS regions and AWS GovCloud (US-East/US-West)", + "tool_support": [ + "Specs (requirements/design/tasks)", + "Agent hooks (event-triggered automations)", + "Steering files", + "MCP servers", + "Terminal CLI with custom agents", + "Checkpoints and property-based spec testing" + ], + "first_release": "Public preview 2025-07-14 (waitlisted within a week due to demand, lifted Oct 2025); GA 2025-11-17. Note: some secondary sources claim a March 2026 startup beta or May 2026 GA — primary sources (kiro.dev, AWS) date GA to 2025-11-17", + "pricing": "Free 50 credits/mo; Pro $20/mo (1,000 credits); Pro+ $40/mo (2,000); Pro Max $100/mo (5,000); Power $200/mo (10,000); overage $0.04/credit; GovCloud ~20% premium, no free tier, IAM Identity Center required", + "company": "Amazon Web Services; Kiro is the designated successor to Amazon Q Developer (Q signups blocked 2026-05-15, end-of-support 2027-04-30)", + "security_incidents": "AWS-2025-019 (2025-10-07): prompt-injection code execution without HITL confirmation, fixed in Kiro 0.1.42; CVE-2026-0830: command injection via workspace folder names in GitLab helper (RCE), fixed in 0.6.18; included in the Dec 2025 'IDEsaster' multi-vendor AI-IDE vulnerability research" + }, + "related_entities": [ + "amazon-bedrock-agents", + "claude-code", + "cursor-agent", + "github-copilot-coding-agent" + ], + "tags": [ + "ide", + "spec-driven", + "aws", + "proprietary" + ] +} diff --git a/data/agents/bytedance-trae.json b/data/agents/bytedance-trae.json new file mode 100644 index 0000000..92ee703 --- /dev/null +++ b/data/agents/bytedance-trae.json @@ -0,0 +1,482 @@ +{ + "id": "bytedance-trae", + "type": "agent", + "name": "Trae", + "provider": "ByteDance", + "version": "2.x (SOLO)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "ByteDance's free AI IDE (VS Code fork) with Builder agent mode and the autonomous SOLO agent (standalone app since March 2026). Aggressive free access to frontier models (Claude, GPT, DeepSeek) drove rapid adoption after its early-2025 launch. Its trust record is the story: 2025 analyses found telemetry continuing after opt-out (~500 network calls in ~7 minutes), persistent hardware-derived device IDs, 5-year post-account data retention, and ByteDance jurisdiction concerns.", + "website": "https://www.trae.ai/", + "trust_vector": { + "performance_reliability": { + "overall_score": 69, + "criteria": { + "task_completion_accuracy": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "InfoQ - ByteDance Launches Trae with DeepSeek R1 and Claude 3.7 Sonnet Free", + "url": "https://www.infoq.com/news/2025/03/trae-bytedance-claude-37-free/", + "date": "2025-03-15", + "value": "Builder mode decomposes prompts into project-level code generation using frontier models (Claude, later GPT and DeepSeek R1) offered free, delivering completion quality comparable to paid competitors" + } + ], + "methodology": "Assessment of Builder/agent output quality given frontier-model backends, from press coverage and user reviews", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Agent loop uses file edits, terminal commands, web search, and MCP servers inside the VS Code-derived IDE; SOLO adds browser and deployment tooling" + } + ], + "methodology": "Review of agent tool loop coverage across editor, terminal, and MCP integrations", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "VibeCoding - Trae Review (2026)", + "url": "https://vibecoding.app/blog/trae-review", + "date": "2026-04-15", + "value": "SOLO mode (launched 2025, standalone desktop/web app 2026-03-31) plans and executes full features end-to-end: requirements, code, test, deploy" + } + ], + "methodology": "Evaluation of SOLO's autonomous plan-build-verify workflow on project-scale tasks", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 65, + "confidence": "low", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Project rules and #Context references persist conventions; long-term cross-session memory is thinner than Cursor's rules/memories system" + } + ], + "methodology": "Review of context and rules persistence features", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 66, + "confidence": "low", + "evidence": [ + { + "source": "VibeCoding - Trae Review (2026)", + "url": "https://vibecoding.app/blog/trae-review", + "date": "2026-04-15", + "value": "Agent iterates on build and lint errors; reviewers report occasional loops and quota-related stalls on free-tier model queues" + } + ], + "methodology": "Assessment of automatic iteration and reported failure modes", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Custom agents with scoped toolsets can be defined, but there is no parallel or background multi-agent orchestration comparable to Cursor 3" + } + ], + "methodology": "Review of custom-agent support versus parallel orchestration capabilities", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 39, + "criteria": { + "tool_sandboxing": { + "score": 55, + "confidence": "low", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Local agent executes terminal commands on the developer machine with approval prompts, in line with IDE-agent norms; SOLO's autonomous mode widens unattended execution" + } + ], + "methodology": "Review of local execution approval model; no isolated-VM option", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 45, + "confidence": "low", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Consumer account model with no SSO, org policy enforcement, or centrally managed privacy controls comparable to enterprise IDE competitors" + } + ], + "methodology": "Review of organizational access governance features", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 45, + "confidence": "low", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Agent ingests untrusted repo content, web search results, and MCP outputs; no published injection-hardening documentation or threat model" + } + ], + "methodology": "Threat surface analysis of untrusted content ingestion against undisclosed mitigations", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 20, + "confidence": "high", + "evidence": [ + { + "source": "Unit 221B - Unveiling Trae: ByteDance's AI IDE and Its Extensive Data Collection System", + "url": "https://blog.unit221b.com/dont-read-this-blog/unveiling-trae-bytedances-ai-ide-and-its-extensive-data-collection-system", + "date": "2025-03-31", + "value": "Continuous transmission to multiple ByteDance servers (mon-va.byteoversea.com, maliva-mcs.byteoversea.com) about every 30 seconds regardless of activity, with complete file contents observed flowing through local WebSocket channels during editing; remote feature-gate system can enable/disable functionality without updates" + }, + { + "source": "The Register - ByteDance AI IDE Trae telemetry continues even after opt-out", + "url": "https://www.theregister.com/2025/07/28/bytedance_trae_telemetry/", + "date": "2025-07-28", + "value": "~500 network calls in ~7 minutes transferring up to 26 MB during active use, continuing after telemetry was disabled; captures machine identifiers, user ID, and project information" + } + ], + "methodology": "Independent network-traffic and binary analyses (Unit 221B March 2025; The Register-covered findings July 2025) of data flows to ByteDance infrastructure", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "The Register - ByteDance AI IDE Trae telemetry continues even after opt-out", + "url": "https://www.theregister.com/2025/07/28/bytedance_trae_telemetry/", + "date": "2025-07-28", + "value": "Closed-source VS Code fork; telemetry behavior had to be reverse-engineered by third parties, and the researcher who reported it was temporarily blocked from Trae's Discord" + } + ], + "methodology": "Source availability and disclosure-culture assessment", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 24, + "criteria": { + "data_retention": { + "score": 15, + "confidence": "high", + "evidence": [ + { + "source": "Trae Privacy Policy", + "url": "https://www.trae.ai/privacy-policy", + "date": "2026-06-01", + "value": "Personal data is retained for five years after the user stops using the service, including post-account closure" + }, + { + "source": "TechRadar - ByteDance AI tool caught collecting user data", + "url": "https://www.techradar.com/pro/security/bytedance-ai-tool-caught-spying-on-users", + "date": "2025-07-30", + "value": "ByteDance confirmed the telemetry toggle only controls VS Code-framework telemetry; collection from other Trae components is unaffected by the opt-out" + } + ], + "methodology": "Privacy policy review: 5-year post-use retention combined with an opt-out that does not stop collection", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 30, + "confidence": "medium", + "evidence": [ + { + "source": "Trae Privacy Policy", + "url": "https://www.trae.ai/privacy-policy", + "date": "2026-06-01", + "value": "Data is shared with ByteDance affiliates and service providers with no clear restrictions on cross-border transfer; access/deletion requests are honored on contact, but demonstrated collection despite opt-out undermines consent validity" + } + ], + "methodology": "Assessment of policy terms and observed behavior against GDPR consent and minimization principles", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 20, + "confidence": "high", + "evidence": [ + { + "source": "Unit 221B - Unveiling Trae: ByteDance's AI IDE and Its Extensive Data Collection System", + "url": "https://blog.unit221b.com/dont-read-this-blog/unveiling-trae-bytedances-ai-ide-and-its-extensive-data-collection-system", + "date": "2025-03-31", + "value": "Telemetry flows to at least five ByteDance domains across Singapore and US endpoints; a hardware-derived SHA-256 machine ID survives reinstallation, enabling long-term cross-session device tracking by the parent company" + } + ], + "methodology": "Network-flow analysis of data destinations and persistent identifiers; ByteDance affiliate sharing is policy-sanctioned", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "The Register - ByteDance AI IDE Trae telemetry continues even after opt-out", + "url": "https://www.theregister.com/2025/07/28/bytedance_trae_telemetry/", + "date": "2025-07-28", + "value": "IDE runs locally but requires cloud model inference and phones home continuously; there is no offline mode, self-hosted option, or effective network-quiet configuration short of DNS-blocking ByteDance domains" + } + ], + "methodology": "Deployment options assessment including feasibility of network isolation", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 53, + "criteria": { + "documentation_quality": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Trae Documentation", + "url": "https://docs.trae.ai/", + "date": "2026-06-01", + "value": "Reasonable product docs for Builder, SOLO, custom agents, rules, and MCP; security and data-flow documentation is minimal and lagged behind third-party findings" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Builder shows step lists, reviewable diffs, and terminal output before applying changes; SOLO streams its plan and actions live" + } + ], + "methodology": "Review of agent action visibility and diff review workflow", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 58, + "confidence": "low", + "evidence": [ + { + "source": "VibeCoding - Trae Review (2026)", + "url": "https://vibecoding.app/blog/trae-review", + "date": "2026-04-15", + "value": "Agent narrates plans and rationale at a level typical of AI IDEs; model routing and quota/queue decisions are not explained" + } + ], + "methodology": "Assessment of plan narration and rationale quality", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 28, + "confidence": "high", + "evidence": [ + { + "source": "Unit 221B - Unveiling Trae: ByteDance's AI IDE and Its Extensive Data Collection System", + "url": "https://blog.unit221b.com/dont-read-this-blog/unveiling-trae-bytedances-ai-ide-and-its-extensive-data-collection-system", + "date": "2025-03-31", + "value": "Closed-source client on an open VS Code base; telemetry framework, feature gates, and agent harness required binary analysis to characterize" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "The Register - ByteDance AI IDE Trae telemetry continues even after opt-out", + "url": "https://www.theregister.com/2025/07/28/bytedance_trae_telemetry/", + "date": "2025-07-28", + "value": "Large, fast-growing user community drawn by free frontier models, but Trae's Discord temporarily blocked the developer who published the telemetry findings — a negative transparency signal" + } + ], + "methodology": "Community size and engagement weighed against vendor handling of critical researchers", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 65, + "criteria": { + "ease_of_integration": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "InfoQ - ByteDance Launches Trae with DeepSeek R1 and Claude 3.7 Sonnet Free", + "url": "https://www.infoq.com/news/2025/03/trae-bytedance-claude-37-free/", + "date": "2025-03-15", + "value": "Familiar VS Code fork importing existing extensions and keybindings, with free frontier-model access from day one — near-zero adoption friction" + } + ], + "methodology": "Onboarding and integration friction assessment", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "VibeCoding - Trae Review (2026)", + "url": "https://vibecoding.app/blog/trae-review", + "date": "2026-04-15", + "value": "Free-tier model access is queue- and quota-limited at peak times; no cloud/background agent fleet for parallel task scaling" + } + ], + "methodology": "Scalability assessment of model quotas and single-agent workflow", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "VibeCoding - Trae Review (2026)", + "url": "https://vibecoding.app/blog/trae-review", + "date": "2026-04-15", + "value": "Generous free tier (premium model access, 5,000 autocompletions/month) with Pro at $10/mo — the cheapest flat pricing among frontier-model AI IDEs" + } + ], + "methodology": "Pricing model analysis; flat low-cost tiers are highly predictable", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 40, + "confidence": "medium", + "evidence": [ + { + "source": "Trae", + "url": "https://www.trae.ai/", + "date": "2026-06-01", + "value": "Per-user quota tracking only; no team dashboards, usage analytics, or audit tooling for organizations" + } + ], + "methodology": "Monitoring and admin features assessment", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "The Register - ByteDance AI IDE Trae telemetry continues even after opt-out", + "url": "https://www.theregister.com/2025/07/28/bytedance_trae_telemetry/", + "date": "2025-07-28", + "value": "ByteDance-scale backing and rapid iteration (SOLO standalone March 2026), but documented telemetry behavior, jurisdiction exposure, and absent enterprise controls make it unsuitable for regulated or sensitive codebases" + } + ], + "methodology": "Maturity assessment weighing vendor scale and cadence against documented trust deficits for professional use", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 74, + "notes": "Capable Builder/SOLO agent with free frontier models; strong value for hobby and non-sensitive projects" + }, + "education": { + "overall": 78, + "notes": "Free access to top models in a familiar IDE is excellent for learners — provided the code and data are not sensitive" + }, + "data-analysis": { + "overall": 58, + "notes": "Fine for writing analysis code, but routing datasets through an IDE with documented exfiltration-pattern telemetry is ill-advised" + }, + "research-assistant": { + "overall": 52, + "notes": "Built-in web search and chat help with technical lookups; not a research product" + } + }, + "best_for": [ + "Hobbyists and students who want free frontier-model agentic coding on non-sensitive projects", + "Budget-constrained developers evaluating AI IDE workflows before paying for alternatives", + "Prototyping and throwaway projects where code confidentiality does not matter" + ], + "not_recommended_for": [ + "Regulated industries (healthcare, finance, government) or any compliance-scoped development", + "Proprietary, sensitive, or client-owned codebases — full file contents were observed in telemetry channels", + "Organizations with data-sovereignty policies restricting ByteDance/PRC-affiliated processors", + "Anyone requiring an effective telemetry opt-out or network-quiet operation", + "Enterprise teams needing SSO, policy enforcement, or audit trails" + ], + "strengths": [ + "Unmatched price-to-capability: free access to Claude, GPT, and DeepSeek R1 class models, Pro at only $10/mo", + "Builder and SOLO agent modes cover the full spectrum from assisted edits to autonomous end-to-end feature delivery", + "Familiar VS Code fork with extension and keybinding compatibility", + "ByteDance-scale infrastructure and rapid release cadence (SOLO standalone app March 2026)", + "Reviewable diffs and step-by-step agent visibility inside the IDE" + ], + "limitations": [ + "Telemetry continued after opt-out: ~500 network calls in ~7 minutes (up to 26 MB) documented in July 2025; ByteDance confirmed the toggle only governs VS Code-framework telemetry", + "Persistent hardware-derived device identifier survives reinstallation, enabling long-term tracking", + "Privacy policy retains personal data for 5 years after use ends and permits sharing with ByteDance affiliates without clear cross-border limits", + "Unit 221B observed complete file contents in local WebSocket telemetry channels and a remote feature-gate system controllable by ByteDance", + "Jurisdiction risk: data flows to ByteDance infrastructure (Singapore/US endpoints) under PRC-affiliated corporate control", + "No enterprise controls (SSO, org policies, audit logs) and no offline/self-hosted mode", + "Vendor transparency record is poor: findings were reverse-engineered by third parties and an early reporter was blocked from the community Discord" + ], + "metadata": { + "license": "Proprietary (VS Code fork)", + "supported_models": [ + "Anthropic Claude", + "OpenAI GPT", + "DeepSeek R1/V3", + "Doubao (ByteDance)" + ], + "languages": [ + "All languages supported by the VS Code ecosystem" + ], + "architecture": "Local VS Code-fork IDE with cloud model inference; Builder agent mode plus autonomous SOLO agent (standalone desktop/web app since 2026-03-31)", + "deployment_type": "Local IDE, cloud inference; no offline or self-hosted option", + "tool_support": [ + "File edits and reviewable diffs", + "Terminal execution", + "Web search", + "MCP servers", + "Custom agents" + ], + "first_release": "January 2025 (international); SOLO mode 2025; SOLO standalone app 2026-03-31", + "pricing": "Free (premium models, 5,000 autocompletions/mo); Pro $10/mo", + "company": "ByteDance (Singapore/US endpoints via byteoversea.com infrastructure; PRC-affiliated parent)" + }, + "related_entities": [ + "cursor-agent", + "github-copilot-coding-agent", + "claude-code", + "openai-codex" + ], + "tags": [ + "ide", + "agentic-coding", + "privacy-risk", + "proprietary" + ] +} diff --git a/data/agents/chatgpt-agent.json b/data/agents/chatgpt-agent.json new file mode 100644 index 0000000..0fed02d --- /dev/null +++ b/data/agents/chatgpt-agent.json @@ -0,0 +1,477 @@ +{ + "id": "chatgpt-agent", + "type": "agent", + "name": "ChatGPT Agent Mode", + "provider": "OpenAI", + "version": "Agent Mode v2 (2026)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Agent mode in ChatGPT that gives the assistant its own virtual computer, merging Operator's web browsing with deep research's analysis. Launched 2025-07-17; Agent Mode v2 (2026) adds persistent memory, scheduled tasks, and GitHub/Jira connectors. Runs in an isolated cloud VM with watch mode for sensitive sites and confirmations before consequential actions, while OpenAI openly acknowledges prompt injection as a core unsolved risk. Included in Plus, Pro, and Team plans with usage caps.", + "website": "https://openai.com/index/introducing-chatgpt-agent/", + "trust_vector": { + "performance_reliability": { + "overall_score": 75, + "criteria": { + "task_completion_accuracy": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI - Introducing ChatGPT agent", + "url": "https://openai.com/index/introducing-chatgpt-agent/", + "date": "2025-07-17", + "value": "Launch model set state-of-the-art results on agentic benchmarks (Humanity's Last Exam, WebArena-class browsing tasks); independent reviews report real-world completion is slower and more variable on complex sites" + }, + { + "source": "4sysops - Testing ChatGPT Agent Mode", + "url": "https://4sysops.com/archives/testing-chatgpt-agent-mode-a-flawed-concept/", + "date": "2025-09-15", + "value": "Independent testing found capable research-plus-action workflows but frequent stalls and confirmation friction on multi-site tasks" + } + ], + "methodology": "Benchmark claims weighed against independent hands-on reviews of end-to-end task completion", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Integrated toolset: visual browser, text browser, terminal with limited network access for code/data/slides, file handling, and first-party connectors (Google Drive, Gmail, GitHub, Jira)" + } + ], + "methodology": "Review of the integrated browser/terminal/connector stack and its reliability across research-and-action workflows", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI - Introducing ChatGPT agent", + "url": "https://openai.com/index/introducing-chatgpt-agent/", + "date": "2025-07-17", + "value": "Fluidly shifts between reasoning and action to handle complex workflows start to finish — the explicit design goal of merging Operator and deep research" + } + ], + "methodology": "Evaluation of long-horizon research-then-act task execution and interruption/resume behavior", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT release notes", + "url": "https://help.openai.com/en/articles/6825453-chatgpt-release-notes", + "date": "2026-06-15", + "value": "Agent Mode v2 (2026) adds persistent memory across agent sessions plus rebuilt scheduled tasks (dedicated Scheduled page, connector access from scheduled runs)" + } + ], + "methodology": "Review of v2 persistent memory, ChatGPT memory integration, and scheduled-task state handling", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 72, + "confidence": "low", + "evidence": [ + { + "source": "4sysops - Testing ChatGPT Agent Mode", + "url": "https://4sysops.com/archives/testing-chatgpt-agent-mode-a-flawed-concept/", + "date": "2025-09-15", + "value": "Retries and re-plans after failed page interactions, but reviewers report loops on CAPTCHA/anti-bot walls and occasional abandoned tasks" + } + ], + "methodology": "Assessment of retry behavior and failure modes reported in independent testing", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 64, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Single-agent design with no user-facing sub-agent composition; delegation exists only via separate scheduled tasks and connectors" + } + ], + "methodology": "Review of multi-agent composition capabilities versus peers with native sub-agents", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 67, + "criteria": { + "tool_sandboxing": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI - Introducing ChatGPT agent", + "url": "https://openai.com/index/introducing-chatgpt-agent/", + "date": "2025-07-17", + "value": "All execution happens on OpenAI's own virtual computer in the cloud — browser, terminal with limited network access, and file tools never touch the user's device" + } + ], + "methodology": "Architecture review of the cloud VM isolation model and restricted terminal network egress", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "ChatGPT agent System Card", + "url": "https://cdn.openai.com/pdf/839e66fc-602c-48bf-81d3-b21eacc3459d/chatgpt_agent_system_card.pdf", + "date": "2025-07-17", + "value": "Explicit user confirmation required before consequential actions (purchases, sending email, editing files); takeover mode lets users enter credentials themselves without the model seeing or storing them; connectors use scoped OAuth subject to workspace admin enablement" + } + ], + "methodology": "Review of confirmation gates, takeover-mode credential handling, and admin controls on connectors", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "ChatGPT agent System Card", + "url": "https://cdn.openai.com/pdf/839e66fc-602c-48bf-81d3-b21eacc3459d/chatgpt_agent_system_card.pdf", + "date": "2025-07-17", + "value": "Layered mitigations: injection-focused fine-tuning, automated monitors that detect and block injection patterns, watch mode requiring active user supervision on sensitive sites (email, banking) that pauses when the user goes inactive, plus internal and external red teaming" + }, + { + "source": "OpenAI - Understanding prompt injections", + "url": "https://openai.com/index/prompt-injections/", + "date": "2025-12-22", + "value": "OpenAI publicly frames prompt injection as a frontier security challenge for its agents that may never be fully solved, committing to continuous adversarial training rather than claiming a fix" + } + ], + "methodology": "Review of documented mitigations and OpenAI's own risk framing; credit for layered defenses and candor, discounted for the acknowledged unsolved core risk across an open-web action surface", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Per-task virtual computer sessions; takeover-mode inputs (e.g., passwords) are not collected or stored; browsing data deletable by the user" + } + ], + "methodology": "Review of session isolation, takeover-mode data handling, and browsing-data deletion controls", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "ChatGPT agent System Card", + "url": "https://cdn.openai.com/pdf/839e66fc-602c-48bf-81d3-b21eacc3459d/chatgpt_agent_system_card.pdf", + "date": "2025-07-17", + "value": "Closed-source product; OpenAI published a detailed system card and a dedicated deployment-safety hub documenting watch mode and mitigations" + } + ], + "methodology": "Source availability assessment; credit for the published system card and safety documentation", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 60, + "criteria": { + "data_retention": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Agent sessions follow ChatGPT retention: consumer chats retained unless deleted with training opt-out available; Business/Enterprise excluded from training by default; users can clear agent browsing data" + } + ], + "methodology": "Review of retention and training-use policies across consumer and workspace tiers as applied to agent sessions", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Trust Portal", + "url": "https://trust.openai.com/", + "date": "2026-03-01", + "value": "SOC 2 Type 2 attestation and GDPR-aligned DPA available for Business/Enterprise workspaces; agent mode inherits these commitments" + } + ], + "methodology": "Compliance certification and DPA availability review under the OpenAI umbrella", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Connectors (Gmail, Google Drive, GitHub, Jira, Slack) and logged-in browsing move data between the agent VM and third-party services; each is user/admin opt-in but broadens flows well beyond OpenAI" + } + ], + "methodology": "Data flow analysis of connector and authenticated-browsing traffic", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI - Introducing ChatGPT agent", + "url": "https://openai.com/index/introducing-chatgpt-agent/", + "date": "2025-07-17", + "value": "Cloud-only: the virtual computer runs exclusively in OpenAI's infrastructure with no self-hosted, on-premises, or offline option" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 68, + "criteria": { + "documentation_quality": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Deployment Safety Hub - ChatGPT agent", + "url": "https://deploymentsafety.openai.com/chatgpt-agent/watch-mode", + "date": "2025-07-17", + "value": "Help center articles, release notes, a full system card, and a deployment-safety hub documenting watch mode, confirmations, and known risks" + } + ], + "methodology": "Documentation completeness review across help center, system card, and safety hub", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI - Introducing ChatGPT agent", + "url": "https://openai.com/index/introducing-chatgpt-agent/", + "date": "2025-07-17", + "value": "Users watch the agent's virtual browser and step narration live, can interrupt or take over at any point; Team plans in v2 add audit log export for scheduled tasks" + } + ], + "methodology": "Review of live session visibility, interruption controls, and workspace audit exports", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Narrates plans and progress and asks before consequential actions, though model/tool selection inside the VM is opaque" + } + ], + "methodology": "Assessment of plan narration and pre-action confirmation clarity", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI - Introducing ChatGPT agent", + "url": "https://openai.com/index/introducing-chatgpt-agent/", + "date": "2025-07-17", + "value": "Fully proprietary agent model and infrastructure; no source availability" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT release notes", + "url": "https://help.openai.com/en/articles/6825453-chatgpt-release-notes", + "date": "2026-06-15", + "value": "Deployed to ChatGPT's massive paid user base with steady feature releases (v2 memory, scheduled tasks, connectors) and extensive third-party coverage and testing" + } + ], + "methodology": "Community engagement analysis via user base scale, release cadence, and public discussion", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 76, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Zero setup: select agent mode from the tools dropdown in any ChatGPT conversation; connectors configured with a few clicks" + } + ], + "methodology": "Onboarding friction assessment for existing ChatGPT subscribers", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Monthly message caps (roughly 400 for Pro, 40 for Plus/Team at launch, with credit-based top-ups); v2 scheduled tasks capped at 25/month for Plus, unlimited on Team" + } + ], + "methodology": "Assessment of usage caps, credit top-ups, and scheduled-task limits across tiers", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT agent", + "url": "https://help.openai.com/en/articles/11752874-chatgpt-agent", + "date": "2026-06-15", + "value": "Included in Plus ($20/mo), Pro ($200/mo), and Team subscriptions — flat pricing with hard caps; optional credits are the only variable spend" + } + ], + "methodology": "Pricing model analysis of subscription-included usage versus optional credit purchases", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center - ChatGPT Business release notes", + "url": "https://help.openai.com/en/articles/11391654-chatgpt-business-release-notes", + "date": "2026-06-15", + "value": "Workspace admins control connector enablement and get audit log export for scheduled agent tasks; per-session visibility for consumers, but no organization-wide agent observability comparable to enterprise agent platforms" + } + ], + "methodology": "Monitoring and admin tooling review across consumer and workspace tiers", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI - Introducing ChatGPT agent", + "url": "https://openai.com/index/introducing-chatgpt-agent/", + "date": "2025-07-17", + "value": "Generally available to paid tiers since 2025-07-17 — nearly a year in production at ChatGPT scale, with a v2 iteration in 2026" + } + ], + "methodology": "Maturity assessment from GA timeline, scale of deployment, and iteration history", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "research-assistant": { + "overall": 85, + "notes": "Strongest fit: deep research heritage plus the ability to act on findings — logging in, refining results, and producing reports, slides, and spreadsheets" + }, + "data-analysis": { + "overall": 76, + "notes": "Terminal in the VM runs code for analysis and generates spreadsheets/slides, though without a persistent data environment" + }, + "content-creation": { + "overall": 74, + "notes": "Produces editable slides and documents from research; formatting fidelity still trails specialist tools" + }, + "code-generation": { + "overall": 60, + "notes": "Can run code and use the v2 GitHub connector, but lacks repository workflows — OpenAI's Codex is the intended coding agent" + } + }, + "best_for": [ + "ChatGPT subscribers delegating research-then-act tasks: comparison shopping, screening, form filling, report building", + "Recurring automated workflows via v2 scheduled tasks with connector access", + "Tasks requiring authenticated web access, using takeover mode to keep credentials away from the model", + "Users who want supervised autonomy with visible browsing and confirmation gates" + ], + "not_recommended_for": [ + "Unsupervised operation on sensitive accounts — watch mode exists precisely because injection risk on email/banking is real", + "High-volume automation (monthly message caps of 40 on Plus/Team)", + "Organizations needing self-hosting or strict data residency" + ], + "strengths": [ + "Clean isolation model: everything executes in OpenAI's cloud virtual computer, never on the user's device", + "Layered, honestly-documented safety: confirmations before consequential actions, watch mode on sensitive sites, injection-trained models and monitors", + "Takeover mode keeps passwords out of the model entirely", + "Merged Operator + deep research design handles research-then-act workflows end to end", + "Agent Mode v2 (2026) added persistent memory, rebuilt scheduled tasks, and GitHub/Jira connectors", + "Included in existing Plus/Pro/Team subscriptions with flat pricing", + "Published system card with internal and external red teaming" + ], + "limitations": [ + "OpenAI itself states prompt injection may never be fully solved; the open-web action surface makes this the defining risk", + "Consumer-grade governance: no org-wide agent observability, limited admin controls outside Business/Enterprise", + "Tight usage caps (roughly 40 messages/month on Plus/Team) constrain real workloads", + "Real-world execution is slow and brittle on complex or bot-protected sites", + "Cloud-only with no self-hosted option; connectors extend data flows to many third parties", + "Single-agent design without sub-agent delegation", + "Closed source with opaque internal tool/model routing" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Dedicated agentic model (o3 family at launch; updated with GPT-5.x generation in 2026)" + ], + "deployment_type": "Cloud-only (per-task virtual computer in OpenAI infrastructure)", + "tool_support": [ + "Visual browser (GUI interaction)", + "Text browser", + "Terminal with limited network access", + "File generation (slides, spreadsheets, documents)", + "Connectors: Google Drive, Gmail, GitHub, Jira, Slack (v2)", + "Scheduled tasks" + ], + "first_release": "2025-07-17 (merger of Operator and deep research); Agent Mode v2 in 2026", + "pricing": "Included in ChatGPT Plus ($20/mo), Pro ($200/mo), and Team; ~400 messages/mo Pro, ~40 Plus/Team, credit top-ups available" + }, + "related_entities": [ + "openai-codex", + "claude-cowork", + "microsoft-scout", + "manus" + ], + "tags": [ + "autonomous", + "general-purpose", + "cloud-agent", + "consumer", + "openai" + ] +} diff --git a/data/agents/claude-cowork.json b/data/agents/claude-cowork.json new file mode 100644 index 0000000..fcbb229 --- /dev/null +++ b/data/agents/claude-cowork.json @@ -0,0 +1,480 @@ +{ + "id": "claude-cowork", + "type": "agent", + "name": "Claude Cowork", + "provider": "Anthropic", + "version": "1.x (desktop GA; web/mobile with cloud execution in beta as of 2026-07-08)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Anthropic's agentic workspace for non-technical knowledge work — 'Claude Code for the office.' Launched 2026-01-12 as a macOS desktop app that runs tasks in a sandboxed local VM (Apple Virtualization Framework with a custom Linux rootfs), expanding to web and mobile with cloud execution in July 2026. Turns natural-language goals into reports, spreadsheets, and organized files. Shipped with a disclosed prompt-injection-via-malicious-files risk that remains its central security tension.", + "website": "https://claude.com/product/cowork", + "trust_vector": { + "performance_reliability": { + "overall_score": 83, + "criteria": { + "task_completion_accuracy": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Claude Cowork expands to mobile and web", + "url": "https://techcrunch.com/2026/07/07/the-coding-agent-wars-are-spilling-into-the-rest-of-the-office-claude-cowork/", + "date": "2026-07-07", + "value": "Analysis of 1.2M anonymized sessions across 600K+ organizations shows sustained production use: 33.4% business-process operations, 16.4% content creation, only 8.7% software development" + } + ], + "methodology": "Assessment of real-world completion using Anthropic's published session analysis and independent launch reviews; built on the proven Claude Code agent harness", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Mature toolset inherited from Claude Code: file manipulation, browser use, Office document generation, connectors (Slack, Google Drive, CRM), and MCP passthrough from the desktop app" + } + ], + "methodology": "Review of tool stack reliability across file, browser, document, and connector operations", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Outcome-based instruction ('what you need, not how'), parallel work streams, scheduled tasks, and unattended execution that continues after the laptop closes" + } + ], + "methodology": "Evaluation of goal decomposition, parallel task execution, and long-horizon unattended runs", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Claude Cowork expands to mobile and web", + "url": "https://techcrunch.com/2026/07/07/the-coding-agent-wars-are-spilling-into-the-rest-of-the-office-claude-cowork/", + "date": "2026-07-07", + "value": "Cross-device continuity: start a task at a desk, get status updates on the phone, retrieve output later; cloud execution keeps sessions alive with no device online" + } + ], + "methodology": "Review of session continuity across desktop, web, and mobile plus workspace file persistence", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 80, + "confidence": "low", + "evidence": [ + { + "source": "PVIEITO - Inside Claude Cowork", + "url": "https://pvieito.com/2026/01/inside-claude-cowork", + "date": "2026-01-15", + "value": "Cowork is Claude Code running inside a VM orchestrated by Claude Desktop, inheriting its self-correcting agent loop; office-task failure modes are less publicly benchmarked than coding" + } + ], + "methodology": "Inference from the shared Claude Code agent loop plus observed retry behavior; limited independent data for non-coding tasks", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Specialized sub-agents handle end-to-end tasks; plugins bundle skills and workflows (brand voice, legal, finance)" + } + ], + "methodology": "Review of sub-agent delegation and plugin/skill composition capabilities", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 65, + "criteria": { + "tool_sandboxing": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "PVIEITO - Inside Claude Cowork", + "url": "https://pvieito.com/2026/01/inside-claude-cowork", + "date": "2026-01-15", + "value": "On macOS, tasks run in a custom Linux VM (Ubuntu-based, ARM64) via Apple's VZVirtualMachine framework with VirtioFS folder sharing, bubblewrap and seccomp inside the VM, per-session sandbox users, and a network allowlist limited to package registries and Anthropic's API" + }, + { + "source": "Claude Help Center - Use Claude Cowork safely", + "url": "https://support.claude.com/en/articles/13364135-use-claude-cowork-safely", + "date": "2026-07-07", + "value": "Cloud sessions run in isolated temporary environments on Anthropic servers that cannot reach home/company networks; however isolation limits where code executes, not what Claude reads or does with granted access" + } + ], + "methodology": "Architecture review of the local VM isolation stack and cloud execution environments; strong containment of code execution, but the allowlisted Anthropic API endpoint has been demonstrated as an exfiltration channel", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Users choose which folders and tools Claude can reach; deletions require approval; enterprise tier adds team-based permissions, spend limits, and per-department tool permissions" + } + ], + "methodology": "Review of folder scoping, approval gates for destructive actions, and enterprise admin controls; consumer defaults rely on users choosing safe folders", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 48, + "confidence": "high", + "evidence": [ + { + "source": "UC Strategies - Anthropic shipped Claude Cowork with a known security flaw", + "url": "https://ucstrategies.com/news/anthropic-shipped-claude-cowork-with-a-known-security-flaw-then-gave-it-to-millions-anyway/", + "date": "2026-01-16", + "value": "Prompt injection via malicious files was disclosed pre-launch (Johann Rehberger) and confirmed within 48 hours of the 2026-01-12 launch: a Word doc with 1-point white text made Cowork upload the user's financial documents to an attacker-controlled Anthropic account via the VM's allowlisted API" + }, + { + "source": "Claude Help Center - Use Claude Cowork safely", + "url": "https://support.claude.com/en/articles/13364135-use-claude-cowork-safely", + "date": "2026-07-07", + "value": "Anthropic documents mitigations (RL training to refuse injected instructions, classifiers scanning untrusted content, deletion protection) but names prompt injection the primary threat and calls agent safety 'an active area of development'; Claude in Chrome and unattended scheduled tasks widen the exposure" + } + ], + "methodology": "Threat analysis of the untrusted-file-plus-consequential-action combination; score reflects a known, demonstrated exploitation path shipped at launch, partially offset by classifiers, training, and honest disclosure", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Help Center - Claude Cowork architecture overview", + "url": "https://support.claude.com/en/articles/14479288-claude-cowork-architecture-overview", + "date": "2026-07-07", + "value": "One VM instance serves multiple conversations, with each conversation in its own isolated session (dedicated sandbox users, bubblewrap); only explicitly shared folders cross the VM boundary" + } + ], + "methodology": "Review of per-session isolation within the shared VM and folder-scoped host access; cloud sessions are per-task and torn down at session end", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 52, + "confidence": "high", + "evidence": [ + { + "source": "Claude Help Center - Claude Cowork architecture overview", + "url": "https://support.claude.com/en/articles/14479288-claude-cowork-architecture-overview", + "date": "2026-07-07", + "value": "Proprietary product; Anthropic publishes unusually detailed architecture and safety documentation, but the agent harness, VM image, and classifiers are closed source" + } + ], + "methodology": "Source availability assessment; credit for published architecture documentation without released code", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 68, + "criteria": { + "data_retention": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic privacy policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-03-01", + "value": "Consumer Pro/Max sessions follow Anthropic consumer retention terms; commercial tiers get training-exclusion defaults and retention controls. Cloud execution moves working files onto Anthropic servers for the session" + } + ], + "methodology": "Review of Anthropic retention commitments applied to Cowork's session files across consumer and enterprise tiers", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-03-01", + "value": "SOC 2 Type II certified with GDPR-aligned DPA available for commercial customers; Cowork enterprise tier adds admin controls and OpenTelemetry activity streaming" + } + ], + "methodology": "Compliance certification and DPA availability review under the Anthropic umbrella", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Processing stays with Anthropic (or customer-selected Bedrock/Google Cloud/Microsoft Foundry hosting); user-enabled connectors (Slack, Drive, CRM, email) move data to those services only when configured" + } + ], + "methodology": "Data flow analysis of model processing and opt-in connector traffic", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 42, + "confidence": "high", + "evidence": [ + { + "source": "Claude Help Center - Claude Cowork architecture overview", + "url": "https://support.claude.com/en/articles/14479288-claude-cowork-architecture-overview", + "date": "2026-07-07", + "value": "Desktop execution happens in a local VM on the user's machine, but Claude models are cloud-only and cloud execution runs entirely on Anthropic servers; no self-hosted or offline option" + } + ], + "methodology": "Deployment options assessment: local execution surface exists but model inference and cloud sessions require Anthropic's cloud", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 73, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Claude Help Center - Get started with Claude Cowork", + "url": "https://support.claude.com/en/articles/13345190-get-started-with-claude-cowork", + "date": "2026-07-07", + "value": "Detailed help center covering setup, safety, and a published architecture overview of the VM isolation stack — unusually candid for a consumer agent" + } + ], + "methodology": "Documentation completeness review including the dedicated safety and architecture articles", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Live progress visibility and plan display before significant actions; enterprise OpenTelemetry activity streaming exists, but Cowork activity 'is not yet captured in audit logs or Compliance API'" + } + ], + "methodology": "Review of session visibility versus enterprise audit coverage; the audit-log gap lowers the score", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Shows its plan and waits for approval before anything significant when permissions are enabled; deletions always require approval" + } + ], + "methodology": "Assessment of plan previews, approval prompts, and progress narration", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 40, + "confidence": "high", + "evidence": [ + { + "source": "PVIEITO - Inside Claude Cowork", + "url": "https://pvieito.com/2026/01/inside-claude-cowork", + "date": "2026-01-15", + "value": "Closed-source product; internals are known mainly through third-party reverse engineering and Anthropic's architecture write-ups" + } + ], + "methodology": "Open source assessment of the agent harness, VM image, and desktop app", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "TechCrunch - Claude Cowork expands to mobile and web", + "url": "https://techcrunch.com/2026/07/07/the-coding-agent-wars-are-spilling-into-the-rest-of-the-office-claude-cowork/", + "date": "2026-07-07", + "value": "1.2M sessions across 600K+ organizations within ~5 months of launch; rapid platform expansion (macOS to Windows/Linux/ChromeOS, web, mobile) and heavy public analysis" + } + ], + "methodology": "Adoption, release cadence, and public discussion volume analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 75, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Included in existing Claude Pro/Max subscriptions with no extra setup; desktop app installs the VM transparently; web and mobile need only a login" + } + ], + "methodology": "Onboarding friction assessment for non-technical users across desktop, web, and mobile", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Claude Cowork expands to mobile and web", + "url": "https://techcrunch.com/2026/07/07/the-coding-agent-wars-are-spilling-into-the-rest-of-the-office-claude-cowork/", + "date": "2026-07-07", + "value": "Cloud execution (beta from 2026-07-07) runs parallel work streams that continue with no device online; five-hour usage windows cap sustained throughput" + } + ], + "methodology": "Assessment of parallel sessions, cloud continuation, and subscription rate-limit behavior", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Flat subscription pricing: included in Pro ($20/mo) and Max ($100-$200/mo); no per-task metering, though Cowork consumes usage limits faster than chat" + } + ], + "methodology": "Pricing model analysis; flat subscriptions cap spend but usage limits can throttle heavy work", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "Claude Cowork product page", + "url": "https://claude.com/product/cowork", + "date": "2026-07-07", + "value": "Enterprise tier offers admin analytics, spend limits, and OpenTelemetry streaming, but Cowork activity is not yet in audit logs or the Compliance API" + } + ], + "methodology": "Monitoring and audit tooling review; consumer tiers have per-session visibility only", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Claude Cowork expands to mobile and web", + "url": "https://techcrunch.com/2026/07/07/the-coding-agent-wars-are-spilling-into-the-rest-of-the-office-claude-cowork/", + "date": "2026-07-07", + "value": "Six months old with fast iteration; web/mobile and cloud execution still in beta; launched with a known prompt-injection flaw, and enterprise audit integration is incomplete" + } + ], + "methodology": "Maturity assessment weighing rapid adoption against beta surfaces and the launch security record", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "content-creation": { + "overall": 84, + "notes": "Core use case: drafts, presentations, proposals, and reports were 16.4% of observed sessions, with document/spreadsheet output as the default deliverable" + }, + "data-analysis": { + "overall": 80, + "notes": "Spreadsheet reconciliation and report assembly dominate the top usage category (33.4% business-process operations)" + }, + "research-assistant": { + "overall": 78, + "notes": "Web browsing plus connector access supports multi-source research organized into documents, with cloud sessions continuing unattended" + }, + "code-generation": { + "overall": 70, + "notes": "Inherits the Claude Code harness and can code (8.7% of sessions), but Claude Code itself is the better tool for engineering work" + } + }, + "best_for": [ + "Non-technical knowledge workers delegating 'work around work': reports, checklists, file organization, spreadsheet reconciliation", + "Finance, HR, and operations roles automating recurring business-process tasks", + "Existing Claude Pro/Max subscribers who want agentic task execution at no extra cost", + "Teams wanting cross-device task continuity with cloud execution that outlives the laptop" + ], + "not_recommended_for": [ + "Processing untrusted documents or emails alongside sensitive files, given the demonstrated file-based prompt-injection exfiltration path", + "Regulated environments needing complete audit trails today (Cowork is not yet in audit logs or the Compliance API)", + "Air-gapped or data-residency-constrained deployments" + ], + "strengths": [ + "Real VM isolation on desktop: custom Linux VM via Apple's VZVirtualMachine with bubblewrap, seccomp, per-session sandbox users, and a network allowlist", + "Built on the proven Claude Code agent harness, repurposed for office work with sub-agents, skills, and plugins", + "Explicit folder scoping and approval gates: Claude can only reach folders the user grants, and deletions require approval", + "Cloud execution (July 2026 beta) continues tasks across desktop, web, and mobile with no device online", + "Unusually candid safety and architecture documentation, including named prompt-injection risks", + "Flat subscription pricing included in Pro/Max removes per-task cost anxiety", + "Massive early traction: 1.2M sessions across 600K+ organizations in ~5 months" + ], + "limitations": [ + "Shipped 2026-01-12 with a pre-disclosed prompt-injection flaw: within 48 hours researchers showed a malicious Word doc (1-pt white text) exfiltrating financial documents via the VM's allowlisted Anthropic API", + "The core tension is architectural: sandboxing limits where code runs, not what Claude does with folders and connectors it was granted", + "Claude in Chrome, computer use, and unattended scheduled tasks each widen the injection-to-action surface", + "Cowork activity not yet captured in enterprise audit logs or the Compliance API", + "Web/mobile and cloud execution still beta with rollout gated by plan tier", + "Consumes subscription usage limits quickly; five-hour usage windows interrupt long tasks", + "Cloud-only models; no self-hosted or offline option" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Claude Opus", + "Claude Sonnet" + ], + "deployment_type": "Desktop app (local VM execution) + web/mobile with Anthropic-hosted cloud execution (beta)", + "architecture": "Claude Code harness inside a custom Linux VM (Apple Virtualization Framework/VZVirtualMachine on macOS; VirtioFS file sharing; bubblewrap + seccomp; network allowlist)", + "tool_support": [ + "File and folder operations (scoped)", + "Browser use / Claude in Chrome", + "Office document, spreadsheet, and presentation generation", + "Connectors (Slack, Google Drive, CRM, email, calendar)", + "MCP passthrough, sub-agents, skills, and plugins", + "Scheduled tasks" + ], + "first_release": "2026-01-12 (macOS desktop, Max then Pro); web/mobile + cloud execution beta 2026-07-07", + "pricing": "Included in Claude Pro ($20/mo) and Max ($100-$200/mo); Team and Enterprise tiers add admin controls", + "usage_stats": "1.2M anonymized sessions across 600K+ organizations (late May 2026): 33.4% business-process ops, 16.4% content creation, 8.7% coding" + }, + "related_entities": [ + "claude-code", + "chatgpt-agent", + "microsoft-scout", + "manus" + ], + "tags": [ + "autonomous", + "knowledge-work", + "consumer", + "sandboxed", + "anthropic" + ] +} diff --git a/data/agents/cline.json b/data/agents/cline.json new file mode 100644 index 0000000..c3f40cc --- /dev/null +++ b/data/agents/cline.json @@ -0,0 +1,479 @@ +{ + "id": "cline", + "type": "agent", + "name": "Cline", + "provider": "Cline Bot Inc.", + "version": "3.x (CLI v3.0.39, 2026-07-09)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "The most-adopted open-source coding agent (5M+ installs by Feb 2026), formerly 'Claude Dev'. Apache-2.0, runs in VS Code, JetBrains, terminal CLI, and via an SDK. Client-side BYOK architecture keeps code local while routing to any model provider; Plan/Act modes with human approval gate edits and commands. Security researchers demonstrated prompt-injection paths that bypassed command approval (mitigated in v3.35.0), and Anthropic's 2026 OAuth crackdown cut off Claude subscription use.", + "website": "https://cline.bot/", + "trust_vector": { + "performance_reliability": { + "overall_score": 79, + "criteria": { + "task_completion_accuracy": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Edits code across projects, runs commands, and uses a browser; 5M+ installs and 64.5K stars indicate sustained real-world task success across model backends" + } + ], + "methodology": "Adoption-signal and hands-on assessment of multi-file coding task completion; accuracy tracks the BYOK model chosen", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Mature file-edit, terminal, and browser tools plus MCP server integration and an MCP marketplace; supports Anthropic, OpenAI, Gemini, OpenRouter (200+ models), Bedrock, Vertex, Ollama, and LM Studio" + } + ], + "methodology": "Tool invocation testing across editors, terminal execution, and MCP integrations", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Signature Plan/Act mode separates strategy from execution: the agent drafts and discusses a plan in read-only mode before switching to approved execution" + } + ], + "methodology": "Evaluation of Plan/Act workflow on long-horizon multi-file changes", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": ".clinerules project files, workspace checkpoints, and the community memory-bank pattern provide cross-session context; no managed memory service" + } + ], + "methodology": "Review of rules files, checkpoints, and task-history persistence", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Monitors terminal output and linter/compiler errors to self-correct; checkpoints let users roll back the workspace to any step" + } + ], + "methodology": "Observed recovery from failing builds and tests plus checkpoint restore testing", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Multi-agent team coordination, scheduled agents, and a headless CLI/SDK for CI pipelines shipped across the 3.x line" + } + ], + "methodology": "Review of multi-agent orchestration features and headless automation support", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 68, + "criteria": { + "tool_sandboxing": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "No OS-level sandbox: commands run in the user's real shell and edits hit the real workspace; safety relies on per-action human approval and configurable auto-approve settings" + } + ], + "methodology": "Execution isolation review; approval gates substitute for sandboxing, and auto-approve weakens them", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Human-in-the-loop approval for every file edit and terminal command by default, with granular auto-approve controls, .clinerules policies, and enterprise controls on team plans" + } + ], + "methodology": "Assessment of the approval model, auto-approve granularity, and enterprise policy features", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Mindgard - Cline coding agent vulnerabilities", + "url": "https://mindgard.ai/blog/cline-coding-agent-vulnerabilities", + "date": "2026-03-25", + "value": "Researchers embedded instructions in Python docstrings, Markdown, and .clinerules to override the requires_approval flag, achieving code execution and data exfiltration from a cloned repo; mitigations including injection detection shipped in v3.35.0" + }, + { + "source": "Adnan Khan - Clinejection", + "url": "https://adnanthekhan.com/posts/clinejection/", + "date": "2026-02-10", + "value": "Prompt injection against Cline's own AI issue-triage workflow could compromise its production release pipeline, demonstrating second-order supply-chain risk" + } + ], + "methodology": "Review of published injection research (approval bypass, TOCTOU payload assembly, supply-chain triager attack) and vendor mitigations", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Cline security documentation", + "url": "https://cline.bot/", + "date": "2026-06-15", + "value": "Client-side architecture: code goes directly from the user's machine to the chosen model provider; Cline's servers do not store or proxy source code in BYOK mode" + } + ], + "methodology": "Data-flow review of the BYOK client-side design and optional Cline-provided inference", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Apache-2.0, fully open extension/CLI/SDK codebase with public issues and rapid security patching (e.g., v3.35.0 injection mitigations)" + } + ], + "methodology": "License and source availability review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 83, + "criteria": { + "data_retention": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Cline website", + "url": "https://cline.bot/", + "date": "2026-06-15", + "value": "Zero code retention by design in BYOK mode: prompts and code flow only to the user's own model provider; telemetry is opt-out" + } + ], + "methodology": "Privacy architecture review of client-side BYOK data flows", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Cline website", + "url": "https://cline.bot/", + "date": "2026-06-15", + "value": "Compliance posture largely inherits from the chosen model provider; team/enterprise plans add organizational controls, but Cline Bot Inc. itself publishes a lighter compliance program than enterprise vendors" + } + ], + "methodology": "Compliance review across BYOK and team-plan configurations", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Code is shared only with the user-selected provider (Anthropic, OpenAI, OpenRouter, local Ollama, etc.); MCP servers the user installs may add further flows" + } + ], + "methodology": "Data-flow analysis across configured providers and user-installed MCP servers", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Supports fully local inference via Ollama and LM Studio and any OpenAI-compatible endpoint, enabling on-prem operation" + } + ], + "methodology": "Deployment options assessment including local-model configurations", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 85, + "criteria": { + "documentation_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Thorough docs for providers, MCP, rules, auto-approve, CLI/SDK, and enterprise deployment" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Every proposed edit is shown as a diff and every command surfaced before execution; task history and checkpoints record the full session" + } + ], + "methodology": "Review of diff previews, task transcripts, and checkpoint history", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Plan mode externalizes the agent's strategy for discussion before any change; per-action approval keeps intent visible throughout" + } + ], + "methodology": "Assessment of plan previews and pre-execution reasoning visibility", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Apache-2.0 with the full extension, CLI, and SDK source public; large fork ecosystem (Roo Code, Kilo Code) built on it" + } + ], + "methodology": "Open source assessment of license and codebase completeness", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "64.5K stars, 6.9K forks, near-daily releases (CLI v3.0.39 on 2026-07-09), active Discord, and 5M+ installs across VS Code, JetBrains, and CLI" + }, + { + "source": "JetBrains Marketplace - Cline plugin", + "url": "https://plugins.jetbrains.com/plugin/28247-cline", + "date": "2026-07-09", + "value": "Native JetBrains plugin extends the ecosystem beyond VS Code" + } + ], + "methodology": "Community engagement analysis via installs, stars, release cadence, and ecosystem forks", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 76, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "One-click marketplace installs for VS Code and JetBrains, npm-installed CLI and SDK; works immediately with any pasted API key" + } + ], + "methodology": "Setup time and integration surface assessment across IDEs, CLI, and SDK", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Headless CLI, SDK, and scheduled agents support CI/CD-scale automation; throughput is bounded by the user's provider rate limits" + } + ], + "methodology": "Assessment of headless/CI usage and multi-instance operation", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Cline website", + "url": "https://cline.bot/", + "date": "2026-06-15", + "value": "Free extension with pay-as-you-go BYOK tokens; heavy full-context usage is a known cost complaint, and Anthropic's OAuth ban removed the cheaper Claude-subscription path in early 2026" + }, + { + "source": "The Register - Anthropic clarifies third-party access ban", + "url": "https://www.theregister.com/2026/02/20/anthropic_clarifies_ban_third_party_claude_access/", + "date": "2026-02-20", + "value": "Anthropic's February 2026 ToS update barred Claude Pro/Max OAuth tokens from third-party tools including Cline; API keys remain the supported route" + } + ], + "methodology": "Pricing analysis of BYOK token costs and the impact of the consumer-OAuth policy change", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Cline documentation", + "url": "https://docs.cline.bot/", + "date": "2026-06-15", + "value": "Per-task token and cost tracking in the UI; team plans add centralized usage visibility, but deep audit/telemetry pipelines require external tooling" + } + ], + "methodology": "Monitoring features assessment for individual and team usage", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Cline GitHub repository", + "url": "https://github.com/cline/cline", + "date": "2026-07-09", + "value": "Mature 3.x line with 5M+ installs (reported Feb 2026), rapid patching of disclosed vulnerabilities, and a stable Plan/Act core that large fork ecosystems depend on" + } + ], + "methodology": "Maturity assessment from adoption scale, release stability, and security response history", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 86, + "notes": "Flagship use case: Plan/Act workflow with diff-gated edits across VS Code, JetBrains, and CLI" + }, + "data-analysis": { + "overall": 76, + "notes": "Handles scripted analysis and notebooks through terminal and file tools" + }, + "research-assistant": { + "overall": 72, + "notes": "Codebase exploration and web research via browser tool and MCP servers" + }, + "content-creation": { + "overall": 66, + "notes": "Fine for technical docs; not built for general content work" + } + }, + "best_for": [ + "Developers wanting a free, open, model-agnostic agent inside their existing IDE", + "Privacy-conscious teams that need code to stay client-side with BYOK routing", + "Users who want explicit human approval of every edit and command (Plan/Act)", + "Teams standardizing agent workflows via .clinerules and the CLI/SDK in CI" + ], + "strengths": [ + "Most-adopted open-source coding agent: 5M+ installs (Feb 2026), 64.5K GitHub stars", + "Client-side BYOK: code never touches Cline servers in default configuration", + "Plan/Act separation with diff previews and per-command approval", + "Very broad provider support: Anthropic, OpenAI, Gemini, OpenRouter (200+ models), Bedrock, Vertex, Ollama, LM Studio", + "Apache-2.0 across extension, CLI, and SDK; spawned a large fork ecosystem", + "MCP marketplace and .clinerules for extensibility and team policy" + ], + "limitations": [ + "No sandbox: approval prompts are the only barrier between the model and the real shell/workspace", + "Demonstrated prompt-injection attacks bypassed command approval via .clinerules/docstrings before v3.35.0 mitigations; auto-approve settings re-open that risk", + "Anthropic ToS change (Jan-Apr 2026) blocked Claude Pro/Max OAuth use, forcing API-key costs on Claude users", + "Full-context prompting makes token costs high on large tasks", + "Compliance program is lighter than enterprise vendors; largely inherited from the chosen provider", + "User-installed MCP servers are an unvetted supply-chain surface" + ], + "metadata": { + "license": "Apache-2.0", + "repository": "https://github.com/cline/cline", + "package_name": "cline (npm CLI); saoudrizwan.claude-dev (VS Code)", + "supported_models": [ + "Anthropic Claude (API key)", + "OpenAI GPT models", + "Google Gemini", + "OpenRouter (200+ models)", + "AWS Bedrock / Azure / GCP Vertex", + "Local via Ollama and LM Studio" + ], + "deployment_type": "Client-side IDE extension (VS Code, JetBrains), terminal CLI, and SDK; BYOK", + "architecture": "Client-side agent with Plan/Act modes, human-in-the-loop approvals, checkpoints, and MCP integration", + "first_release": "July 2024 (as Claude Dev); renamed Cline early 2025", + "adoption": "5M+ installs across VS Code, JetBrains, and CLI (reported February 2026)", + "pricing": "Free extension + BYOK provider costs; paid team/enterprise plans with centralized billing and controls", + "github_stars": "64500+" + }, + "related": [ + "claude-code", + "cursor-agent", + "opencode", + "github-copilot-coding-agent", + "openai-codex" + ], + "tags": [ + "coding-agent", + "open-source", + "byok", + "ide-extension", + "mcp" + ] +} diff --git a/data/agents/factory-droids.json b/data/agents/factory-droids.json new file mode 100644 index 0000000..4d841a2 --- /dev/null +++ b/data/agents/factory-droids.json @@ -0,0 +1,474 @@ +{ + "id": "factory-droids", + "type": "agent", + "name": "Factory Droids", + "provider": "Factory AI", + "version": "Droids platform (2026, Missions era)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Factory AI's enterprise agent-native software development platform. Specialist autonomous Droids handle coding, testing, code review, refactoring, and DevOps work across Desktop, CLI, SDK, Slack, ticketing, and CI, with Missions enabling long-horizon multi-agent workflows. #1 on Terminal-Bench (Sept 2025); $220M raised at a $1.5B valuation (Apr 2026); used daily by hundreds of thousands of developers at enterprises including Nvidia, Adobe, and EY.", + "website": "https://factory.ai/", + "trust_vector": { + "performance_reliability": { + "overall_score": 82, + "criteria": { + "task_completion_accuracy": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Factory - Droid #1 on Terminal-Bench", + "url": "https://factory.ai/news/terminal-bench", + "date": "2025-09-25", + "value": "Droid reached #1 on Terminal-Bench at 58.8%, outperforming Claude Code and Codex CLI; Factory harnesses held three of the top five slots (Opus 4.1 58.8%, GPT-5 52.5%, Sonnet 4 50.5%), showing harness design drives results across models" + }, + { + "source": "Factory leaderboards documentation", + "url": "https://docs.factory.ai/leaderboards", + "date": "2026-07-09", + "value": "Factory maintains published leaderboard results and claims #1 status across leading software development agent benchmarks" + } + ], + "methodology": "Review of public Terminal-Bench leaderboard placement (product-harness category) and sustained benchmark results; vendor-run submissions weighted with independent leaderboard verification", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Factory Series C announcement", + "url": "https://factory.ai/news/series-c", + "date": "2026-04-16", + "value": "Droids operate across terminal, editor, Slack, ticketing, and CI with full system access and local context via Factory Desktop; works with any model and any interface" + } + ], + "methodology": "Assessment of multi-surface tool execution (shell, editor, integrations) and MCP/tool configuration reliability", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Factory Series C announcement", + "url": "https://factory.ai/news/series-c", + "date": "2026-04-16", + "value": "Missions capability enables long-horizon, multi-step, multi-agent workflows spanning stages of the SDLC" + } + ], + "methodology": "Evaluation of Missions long-horizon planning and spec-driven execution across multi-stage engineering tasks", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Factory documentation", + "url": "https://docs.factory.ai/", + "date": "2026-07-09", + "value": "Organization-level context: AGENTS.md project guidance, agent-readiness dashboard, and persistent org/repo context shared across Droids and sessions" + } + ], + "methodology": "Review of org context engineering, project guidance files, and cross-session context persistence", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Factory - Droid #1 on Terminal-Bench", + "url": "https://factory.ai/news/terminal-bench", + "date": "2025-09-25", + "value": "Terminal-Bench tasks include debugging, legacy modernization, and build/test loops where Droid's harness recovered from failures better than competing agents even on sub-frontier models" + } + ], + "methodology": "Inference from benchmark categories requiring iterative failure recovery plus documented test-iteration behavior", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Factory Series C announcement", + "url": "https://factory.ai/news/series-c", + "date": "2026-04-16", + "value": "Specialist Droids (coding, review, testing, docs, DevOps-style tasks) coordinate in multi-agent Missions; work is delegated from Slack, tickets, or CI and returns as reviewable output" + } + ], + "methodology": "Review of specialist-Droid division of labor, Missions multi-agent orchestration, and delegation surfaces", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 69, + "criteria": { + "tool_sandboxing": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Factory CLI security documentation", + "url": "https://docs.factory.ai/cli/account/security", + "date": "2026-07-09", + "value": "Shell commands and file edits execute locally with only necessary context/diffs sent to Factory's cloud; optional OS-level sandbox with kernel-enforced filesystem/network isolation (Beta); Droid Computers provide Factory-managed cloud sandboxes on Plus/Max" + } + ], + "methodology": "Architecture review of local execution scoping, beta OS-level sandbox, and managed cloud sandbox model", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Factory CLI security documentation", + "url": "https://docs.factory.ai/cli/account/security", + "date": "2026-07-09", + "value": "SAML 2.0/OIDC SSO with SCIM provisioning, role-based access controls, per-tool allow/ask/reject permissions, write access restricted to project directories, and enterprise-managed security policies" + } + ], + "methodology": "Review of identity, RBAC, tool-permission granularity, and centrally managed policy controls", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Factory CLI security documentation", + "url": "https://docs.factory.ai/cli/account/security", + "date": "2026-07-09", + "value": "Prompt injection detection and input sanitization documented as built-in, with risky operations requiring explicit user approval; effectiveness not yet validated by third-party research" + } + ], + "methodology": "Review of documented injection mitigations (rare among peers) tempered by absence of independent adversarial testing", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Factory CLI security documentation", + "url": "https://docs.factory.ai/cli/account/security", + "date": "2026-07-09", + "value": "TLS 1.3 in transit, AES-256 at rest with AWS KMS, customer-managed encryption keys (BYOK), zero data retention mode, and dedicated compute with partitioned inference on Enterprise" + } + ], + "methodology": "Review of encryption, key management, ZDR, and tenant partitioning claims", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Factory documentation", + "url": "https://docs.factory.ai/", + "date": "2026-07-09", + "value": "Fully proprietary platform and harness; no source code published, though docs, leaderboard methodology, and a public trust center (trust.factory.ai) provide partial transparency" + } + ], + "methodology": "Source availability assessment", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 74, + "criteria": { + "data_retention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Factory CLI security documentation", + "url": "https://docs.factory.ai/cli/account/security", + "date": "2026-07-09", + "value": "Explicit commitment that Factory never trains on customer code; Zero Data Retention mode included on Business tier and above" + } + ], + "methodology": "Review of no-training commitment and ZDR tier availability", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Factory CLI security documentation", + "url": "https://docs.factory.ai/cli/account/security", + "date": "2026-07-09", + "value": "SOC 2 Type II certified with regular penetration testing and GDPR-compliant operations; compliance documentation via trust.factory.ai" + } + ], + "methodology": "Compliance certification review (SOC 2 Type II, GDPR) via security docs and trust center", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Factory pricing", + "url": "https://factory.ai/pricing", + "date": "2026-07-09", + "value": "Workloads route to third-party frontier models (GPT-5, Claude Opus/Sonnet, Gemini); Enterprise offers dedicated compute with partitioned inference to tighten the data path" + } + ], + "methodology": "Data flow analysis of multi-provider model routing and enterprise inference isolation options", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Factory pricing", + "url": "https://factory.ai/pricing", + "date": "2026-07-09", + "value": "Enterprise tier offers on-premise deployment, data residency options, and customer-managed encryption keys; CLI executes locally by default, but the platform and inference are cloud services" + } + ], + "methodology": "Deployment options assessment: on-prem/private-cloud available at Enterprise, no self-contained air-gapped offering", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 66, + "criteria": { + "documentation_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Factory documentation", + "url": "https://docs.factory.ai/", + "date": "2026-07-09", + "value": "Thorough docs spanning CLI, security, enterprise administration, leaderboard methodology, and integration setup" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Factory CLI security documentation", + "url": "https://docs.factory.ai/cli/account/security", + "date": "2026-07-09", + "value": "Complete session logging maintained with OpenTelemetry metrics integration and enterprise audit-oriented administration" + } + ], + "methodology": "Review of session logs, OTel telemetry, and audit features for governance", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Factory Series C announcement", + "url": "https://factory.ai/news/series-c", + "date": "2026-04-16", + "value": "Droids produce reviewable plans, diffs, and PR-style outputs; Missions expose multi-step progress, though internal harness decisions are less visible than open competitors" + } + ], + "methodology": "Assessment of plan/diff visibility and change justification in delegated workflows", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "Factory documentation", + "url": "https://docs.factory.ai/", + "date": "2026-07-09", + "value": "Closed-source product; no public source repositories for the platform or Droid harness" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Factory Series C announcement", + "url": "https://factory.ai/news/series-c", + "date": "2026-04-16", + "value": "Hundreds of thousands of daily developers and revenue doubling month-over-month for six months; community is enterprise-customer-driven rather than open-source contributor-driven" + } + ], + "methodology": "Community and ecosystem engagement analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 81, + "criteria": { + "ease_of_integration": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Factory Series C announcement", + "url": "https://factory.ai/news/series-c", + "date": "2026-04-16", + "value": "Droids accessible from Desktop, CLI, and SDK, plus Slack, ticketing systems, and CI; works with any model and existing developer tooling" + } + ], + "methodology": "Integration surface assessment across developer and team-workflow entry points", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Factory pricing", + "url": "https://factory.ai/pricing", + "date": "2026-07-09", + "value": "Droid Computers provide Factory-managed cloud sandboxes for parallel background agents; Enterprise supports unlimited members with dedicated compute; proven at Nvidia, Adobe, EY, Palo Alto Networks, Adyen scale" + } + ], + "methodology": "Assessment of parallel cloud execution, enterprise seat scaling, and named large-scale deployments", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Factory pricing", + "url": "https://factory.ai/pricing", + "date": "2026-07-09", + "value": "Individual tiers Pro $20/mo, Plus $100/mo (~5x usage), Max $200/mo (~10x usage); Business/Enterprise custom-priced with tailored usage limits per seat. Usage-based consumption within tiers varies with task size" + } + ], + "methodology": "Pricing model analysis: flat tiers with opaque underlying usage allowances; enterprise seat+usage blend requires negotiation", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Factory pricing", + "url": "https://factory.ai/pricing", + "date": "2026-07-09", + "value": "Billing tracking and agent-readiness dashboards on all paid tiers; session logging, OpenTelemetry metrics, and enterprise admin controls for governance" + } + ], + "methodology": "Review of usage dashboards, telemetry export, and admin governance tooling", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Factory Series C announcement", + "url": "https://factory.ai/news/series-c", + "date": "2026-04-16", + "value": "$150M Series C led by Khosla Ventures (Sequoia, Blackstone, Insight, NEA participating) at $1.5B valuation; $220M raised in total; revenue doubling monthly for six months with major enterprise customers" + } + ], + "methodology": "Vendor viability and deployment maturity assessment from funding, revenue trajectory, and named enterprise adoption", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 86, + "notes": "Terminal-Bench leader; specialist Droids cover coding, review, testing, and refactoring with strong results even on sub-frontier models" + }, + "data-analysis": { + "overall": 68, + "notes": "Can script analyses and ML workflows in its environments, but the platform targets the software delivery lifecycle" + }, + "research-assistant": { + "overall": 64, + "notes": "Codebase and ticket-context research is strong; general-purpose research is not the product's aim" + }, + "customer-service": { + "overall": 40, + "notes": "Not designed for customer-facing conversational work; Slack presence is for engineering delegation only" + } + }, + "best_for": [ + "Enterprises wanting governed, auditable agents across the full SDLC (code, test, review, deploy)", + "Teams delegating engineering work from Slack, Jira/ticketing, and CI rather than an IDE", + "Organizations with strict procurement needs: SOC 2 Type II, ZDR, BYOK keys, on-prem options", + "Large-scale parallel agent execution via Factory-managed Droid Computers", + "Model-flexible shops running GPT-5, Claude, and Gemini through one harness" + ], + "strengths": [ + "#1 on Terminal-Bench (58.8%, Sept 2025) with three of the top five harness slots, beating Claude Code and Codex CLI", + "Specialist Droids plus Missions for long-horizon, multi-step, multi-agent SDLC workflows", + "Strong enterprise security package: SOC 2 Type II, SSO/SCIM, RBAC, allow/ask/reject tool permissions, ZDR, BYOK keys, on-prem", + "Documented prompt injection detection and input sanitization - rare among coding agents", + "Observability for governance: complete session logs, OpenTelemetry metrics, admin dashboards", + "Proven enterprise adoption: Nvidia, Adobe, EY, Palo Alto Networks, Adyen; hundreds of thousands of daily developers", + "Strong vendor trajectory: $220M raised, $1.5B valuation (Apr 2026), revenue doubling monthly for six months" + ], + "limitations": [ + "Fully closed-source platform and harness; transparency relies on vendor documentation", + "OS-level sandbox is still Beta; default local execution relies on permission gates", + "Usage allowances behind flat tiers are opaque, making heavy-use costs hard to forecast", + "Injection defenses and security claims lack independent third-party validation", + "Workloads depend on third-party frontier model providers (GPT-5, Claude, Gemini)", + "Rapid growth-stage company: pricing, packaging, and product surface have churned repeatedly since 2025", + "Best experience requires adopting Factory's platform surfaces; deep vendor coupling for SDLC-wide automation" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "OpenAI GPT-5", + "Anthropic Claude Opus and Sonnet", + "Google Gemini", + "Additional models via harness (model-agnostic design)" + ], + "programming_languages": [ + "Most major languages (Python, TypeScript, Java, Go, C++, etc.)" + ], + "deployment_type": "Cloud platform + local CLI/Desktop execution; Droid Computers managed cloud sandboxes; on-premise available at Enterprise", + "tool_support": [ + "Droid CLI, Desktop, and SDK", + "Slack and ticketing (Jira/Linear-style) delegation", + "CI/CD integration", + "MCP and custom tools", + "OpenTelemetry metrics" + ], + "first_release": "Founded 2023; Droid CLI GA and Terminal-Bench #1 September 2025; Missions/Series C April 2026", + "pricing": "Individual: Pro $20/mo, Plus $100/mo, Max $200/mo (usage-tiered); Business (up to 150 seats, ZDR, SSO) and Enterprise (unlimited, dedicated compute, on-prem) custom seat+usage pricing", + "company_milestones": "$220M total raised; $150M Series C at $1.5B valuation led by Khosla Ventures (2026-04-16); revenue doubled month-over-month for six consecutive months; customers include Nvidia, Adobe, EY, Palo Alto Networks, Adyen" + }, + "related_entities": [ + "devin", + "claude-code", + "openai-codex", + "github-copilot-coding-agent", + "cursor-agent" + ], + "tags": [ + "autonomous", + "enterprise", + "sdlc-platform", + "multi-agent", + "proprietary" + ] +} diff --git a/data/agents/google-antigravity.json b/data/agents/google-antigravity.json new file mode 100644 index 0000000..94149b0 --- /dev/null +++ b/data/agents/google-antigravity.json @@ -0,0 +1,481 @@ +{ + "id": "google-antigravity", + "type": "agent", + "name": "Google Antigravity", + "provider": "Google", + "version": "Public preview (2026)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Google's agent-first development platform (launched 2025-11-18 with Gemini 3): an IDE built on licensed Windsurf code where an agent manager orchestrates autonomous agents across editor, terminal, and browser, verifying work through artifacts (plans, task lists, screenshots, recordings). Also the home of the closed-source 'agy' CLI that replaced consumer Gemini CLI (June 2026). Free public preview whose first months were marred by serious, well-documented security failures.", + "website": "https://antigravity.google/", + "trust_vector": { + "performance_reliability": { + "overall_score": 76, + "criteria": { + "task_completion_accuracy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Google Developers Blog - Build with Google Antigravity", + "url": "https://developers.googleblog.com/build-with-google-antigravity-our-new-agentic-development-platform/", + "date": "2025-11-18", + "value": "Agents powered by Gemini 3 Pro (plus Claude Sonnet/Opus and GPT-OSS-120B options) complete asynchronous coding tasks end-to-end, with browser control for verifying web apps" + } + ], + "methodology": "Assessment of agentic task completion using Gemini 3-class models and multi-surface tooling, from vendor documentation and independent reviews", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "VentureBeat - Antigravity agent-first architecture", + "url": "https://venturebeat.com/ai/google-antigravity-introduces-agent-first-architecture-for-asynchronous", + "date": "2025-11-18", + "value": "Agents operate editor, terminal, and an embedded browser in one loop; browser-use verification is a differentiator, though preview-quality rough edges are widely reported" + } + ], + "methodology": "Review of cross-surface (editor/terminal/browser) tool loop reliability in public preview", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Google Developers Blog - Build with Google Antigravity", + "url": "https://developers.googleblog.com/build-with-google-antigravity-our-new-agentic-development-platform/", + "date": "2025-11-18", + "value": "Agents produce implementation plans and task lists as first-class artifacts before executing, designed for delegating complex multi-step work" + } + ], + "methodology": "Evaluation of plan-artifact generation and adherence on long-horizon tasks", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 70, + "confidence": "low", + "evidence": [ + { + "source": "Google Antigravity", + "url": "https://antigravity.google/", + "date": "2026-06-01", + "value": "Agents accumulate a knowledge base of learnings across tasks within the workspace; persistence depth across projects is not well documented in preview" + } + ], + "methodology": "Review of cross-task knowledge retention features as documented in preview materials", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 72, + "confidence": "low", + "evidence": [ + { + "source": "VentureBeat - Antigravity agent-first architecture", + "url": "https://venturebeat.com/ai/google-antigravity-introduces-agent-first-architecture-for-asynchronous", + "date": "2025-11-18", + "value": "Agents iterate against test results and browser-verified behavior; preview users report occasional runaway or stalled tasks requiring manual intervention" + } + ], + "methodology": "Assessment of autonomous iteration and failure handling from launch coverage and user reports", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Google Developers Blog - Build with Google Antigravity", + "url": "https://developers.googleblog.com/build-with-google-antigravity-our-new-agentic-development-platform/", + "date": "2025-11-18", + "value": "The Manager view is a mission-control surface for spawning and supervising multiple agents working in parallel across workspaces" + } + ], + "methodology": "Review of the agent manager orchestration model for parallel task execution", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 42, + "criteria": { + "tool_sandboxing": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "Pillar Security - Prompt Injection leads to RCE and Sandbox Escape in Antigravity", + "url": "https://www.pillar.security/blog/prompt-injection-leads-to-rce-and-sandbox-escape-in-antigravity", + "date": "2026-04-20", + "value": "Critical flaw: the find_by_name tool passed its Pattern parameter unsanitized to fd, so injecting the -X (exec-batch) flag executed arbitrary binaries, bypassing Strict Mode entirely because the tool ran before sandbox constraints were evaluated. Reported 2026-01-07, fixed 2026-02-28, disclosed 2026-04-20; triggerable via indirect prompt injection. No CVE assigned" + }, + { + "source": "The Hacker News - Google Patches Antigravity IDE Flaw", + "url": "https://thehackernews.com/2026/04/google-patches-antigravity-ide-flaw.html", + "date": "2026-04-21", + "value": "Google patched the sandbox-escape RCE; Strict Mode's guarantees (network limits, no out-of-workspace writes, sandboxed commands) were shown to be bypassable by a single crafted tool parameter" + } + ], + "methodology": "Security review of the sandbox model against the published bypass; the flaw is patched, but a sandbox defeated by flag injection in a core tool indicates immature enforcement architecture", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "TechRadar Pro - Antigravity's worrying security issues", + "url": "https://www.techradar.com/pro/googles-ai-powered-antigravity-ide-already-has-some-worrying-security-issues", + "date": "2025-11-26", + "value": "At launch, default policies ('Agent Decides' review mode, terminal command auto-execution) let agents run commands and read credentials without per-action approval; consumer preview lacks enterprise admin controls" + } + ], + "methodology": "Review of default autonomy policies, approval model, and org-level controls in the preview", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 28, + "confidence": "high", + "evidence": [ + { + "source": "PromptArmor - Google Antigravity Exfiltrates Data", + "url": "https://www.promptarmor.com/resources/google-antigravity-exfiltrates-data", + "date": "2025-11-25", + "value": "Launch-week demonstration: hidden instructions in a poisoned integration guide made the agent collect .env credentials and exfiltrate them via the browser subagent to webhook.site, which sat on the default URL allowlist" + }, + { + "source": "Simon Willison - Google Antigravity Exfiltrates Data", + "url": "https://simonwillison.net/2025/Nov/25/google-antigravity-exfiltrates-data/", + "date": "2025-11-25", + "value": "The agent bypassed its own .gitignore protection by cat-ing files through run_command; Google's Bug Hunters page classifies prompt-injection data exfiltration and code execution as 'known issues' ineligible for bounty, several inherited from the Windsurf codebase and reported to Windsurf in May 2025" + } + ], + "methodology": "Assessment of demonstrated indirect prompt injection attack chains at launch against documented mitigations; multiple independent researchers achieved credential exfiltration with default settings", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 45, + "confidence": "medium", + "evidence": [ + { + "source": "Google Bug Hunters - Antigravity Known Issues", + "url": "https://bughunters.google.com/learn/invalid-reports/ai-products/antigravity-known-issues", + "date": "2025-11-25", + "value": "Google acknowledges data exfiltration via the browser agent and prompt-injection code execution as known issues under remediation; workspace secrets and local files are reachable by agents by design" + }, + { + "source": "Malwarebytes - Fake Google Antigravity downloads", + "url": "https://www.malwarebytes.com/blog/threat-intel/2026/04/fake-google-antigravity-downloads-are-stealing-accounts-in-minutes", + "date": "2026-04-15", + "value": "Supply-chain adjacent risk: malvertised lookalike domains shipped the genuine Antigravity installer trojanized with an infostealer, harvesting browser sessions and credentials within minutes" + } + ], + "methodology": "Review of agent access to local secrets, exfiltration channels, and the ecosystem attack surface around distribution", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 40, + "confidence": "high", + "evidence": [ + { + "source": "The New Stack - Gemini CLI vs Antigravity", + "url": "https://thenewstack.io/gemini-cli-antigravity-replacement/", + "date": "2026-06-20", + "value": "Closed-source VS Code fork built on non-exclusively licensed Windsurf code; the agy CLI is a closed Go binary that replaced the Apache-2.0 Gemini CLI (105K+ stars, 6,000+ community PRs). Partial credit: Google publicly documents known security issues and paid Pillar's bounty" + } + ], + "methodology": "Source availability assessment, crediting the public known-issues documentation while noting the open-to-closed regression from Gemini CLI", + "last_verified": "2026-07-09" + } + }, + "notes": "Security is the defining weakness of Antigravity's first six months: a Strict Mode sandbox escape to RCE (reported 2026-01-07, patched 2026-02-28), launch-day credential exfiltration via indirect prompt injection with default-on auto-execution, inherited unpatched Windsurf issues, and an active trojanized-installer campaign. Google's pedigree did not translate into a hardened launch." + }, + "privacy_compliance": { + "overall_score": 52, + "criteria": { + "data_retention": { + "score": 45, + "confidence": "medium", + "evidence": [ + { + "source": "Google Antigravity Terms", + "url": "https://antigravity.google/terms", + "date": "2026-06-01", + "value": "Preview terms allow Google to use interactions (prompts, code context) to improve and train models by default for consumer accounts, with an opt-out; Workspace/GCP-governed accounts are excluded from training use" + } + ], + "methodology": "Review of preview terms of service regarding training use and retention of interaction data", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Google Antigravity", + "url": "https://antigravity.google/", + "date": "2026-06-01", + "value": "Backed by Google's corporate compliance programs (GDPR, ISO/SOC portfolio), but Antigravity itself is a free public preview without product-specific compliance attestations or enterprise data terms" + } + ], + "methodology": "Compliance posture assessment distinguishing Google-level programs from preview-product guarantees", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Google Developers Blog - Build with Google Antigravity", + "url": "https://developers.googleblog.com/build-with-google-antigravity-our-new-agentic-development-platform/", + "date": "2025-11-18", + "value": "Model choice includes Anthropic Claude Sonnet/Opus and GPT-OSS-120B alongside Gemini, routing code and prompts to non-Google providers when selected" + } + ], + "methodology": "Data flow analysis of multi-provider model routing", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 42, + "confidence": "high", + "evidence": [ + { + "source": "Google Antigravity", + "url": "https://antigravity.google/", + "date": "2026-06-01", + "value": "IDE and agents run locally on macOS/Windows/Linux, but all model inference is cloud-based with a Google account requirement; no offline or self-hosted inference" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 61, + "criteria": { + "documentation_quality": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Google Antigravity", + "url": "https://antigravity.google/", + "date": "2026-06-01", + "value": "Launch documentation covers the manager surface, artifacts, and model options; security-relevant defaults and policies were underdocumented at launch relative to the product's autonomy" + } + ], + "methodology": "Documentation completeness review for a preview-stage product", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "VentureBeat - Antigravity agent-first architecture", + "url": "https://venturebeat.com/ai/google-antigravity-introduces-agent-first-architecture-for-asynchronous", + "date": "2025-11-18", + "value": "Artifact-based verification is the product's core idea: agents emit task lists, implementation plans, screenshots, and browser recordings as a reviewable audit trail instead of raw tool-call logs" + } + ], + "methodology": "Review of the artifact audit-trail model; genuinely strong design, though artifacts are agent-generated and share the self-reporting caveat", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Google Developers Blog - Build with Google Antigravity", + "url": "https://developers.googleblog.com/build-with-google-antigravity-our-new-agentic-development-platform/", + "date": "2025-11-18", + "value": "Implementation-plan artifacts expose intended approach before execution, and verification artifacts (screenshots, recordings) evidence what was actually done" + } + ], + "methodology": "Assessment of plan and verification artifact quality as explanation mechanisms", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 22, + "confidence": "high", + "evidence": [ + { + "source": "9to5Google - Gemini CLI and Code Assist shutting down for consumers", + "url": "https://9to5google.com/2026/06/17/gemini-cli-code-assist-shutting-down/", + "date": "2026-06-17", + "value": "Closed-source platform; Google shut off the open-source Gemini CLI for consumer tiers on 2026-06-18 and pointed individuals to the closed agy binary, a full rewrite shipping ~20 free requests/day versus Gemini CLI's 1,000" + } + ], + "methodology": "Open source assessment; the deliberate replacement of a major OSS tool with a closed binary weighs against Google here", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "The New Stack - Gemini CLI vs Antigravity", + "url": "https://thenewstack.io/gemini-cli-antigravity-replacement/", + "date": "2026-06-20", + "value": "High adoption interest from the Gemini 3 launch wave, but significant community backlash over the Gemini CLI shutdown, free-tier quota cuts, and the closed-source pivot" + } + ], + "methodology": "Community engagement analysis balancing launch momentum against migration backlash", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 63, + "criteria": { + "ease_of_integration": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Google Developers Blog - Build with Google Antigravity", + "url": "https://developers.googleblog.com/build-with-google-antigravity-our-new-agentic-development-platform/", + "date": "2025-11-18", + "value": "Free download for macOS/Windows/Linux with a familiar VS Code-derived editor, generous Gemini 3 Pro rate limits, and sign-in with a Google account" + } + ], + "methodology": "Onboarding and integration friction assessment", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 70, + "confidence": "low", + "evidence": [ + { + "source": "Google Antigravity", + "url": "https://antigravity.google/", + "date": "2026-06-01", + "value": "Manager view fans out parallel agents locally; scale is bounded by preview rate limits, which Google adjusts without an SLA" + } + ], + "methodology": "Scalability assessment of parallel agent execution under preview rate limits", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "9to5Google - Gemini CLI and Code Assist shutting down for consumers", + "url": "https://9to5google.com/2026/06/17/gemini-cli-code-assist-shutting-down/", + "date": "2026-06-17", + "value": "Currently free, but preview quotas shift without notice, agy's free tier is roughly 20 requests/day, and post-preview pricing is unannounced; the Gemini CLI shutdown shows Google will change individual-tier terms abruptly" + } + ], + "methodology": "Pricing predictability analysis; zero cost today but high uncertainty about limits and future pricing", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 58, + "confidence": "low", + "evidence": [ + { + "source": "Google Antigravity", + "url": "https://antigravity.google/", + "date": "2026-06-01", + "value": "Manager view provides per-agent progress and artifacts, but there are no organizational dashboards, usage analytics, or admin policy controls in the consumer preview" + } + ], + "methodology": "Monitoring and governance features assessment for the preview release", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "Dark Reading - Google Fixes Critical RCE Flaw in Antigravity", + "url": "https://www.darkreading.com/vulnerabilities-threats/google-fixes-critical-rce-flaw-ai-based-antigravity-tool", + "date": "2026-04-21", + "value": "Public preview with an eight-week-old critical RCE at disclosure, unresolved 'known issue' prompt-injection classes, forced consumer migration churn from Gemini CLI, and no enterprise tier; Google's backing ensures longevity but not current stability" + } + ], + "methodology": "Product maturity assessment weighing Google's resources against preview status and the 2025-2026 security record", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 78, + "notes": "Strong Gemini 3-powered agentic coding with a genuinely novel artifact-verification workflow, held back by preview instability and security caveats" + }, + "research-assistant": { + "overall": 66, + "notes": "Embedded browser control lets agents research documentation and verify behavior, but browsing is also its most-exploited attack surface" + }, + "data-analysis": { + "overall": 62, + "notes": "Capable of building and running analysis code across editor and terminal; no analytics-specific tooling" + }, + "education": { + "overall": 64, + "notes": "Free access and visible plans/artifacts aid learning, but default autonomy settings are a poor fit for unsupervised beginners" + } + }, + "best_for": [ + "Developers experimenting with agent-manager workflows and artifact-based verification at no cost", + "Gemini-ecosystem teams evaluating Google's agentic direction before enterprise tooling matures", + "Multi-agent parallel task delegation on non-sensitive codebases", + "Web app development where browser-based verification of agent work adds value" + ], + "strengths": [ + "Artifact-based verification (plans, task lists, screenshots, browser recordings) is a best-in-class transparency design", + "Agent manager orchestrates parallel autonomous agents across editor, terminal, and browser in one surface", + "First-class Gemini 3 Pro access with model choice (Claude Sonnet/Opus, GPT-OSS-120B) at zero cost in preview", + "Google patched the critical sandbox-escape RCE within eight weeks of report and paid an external bounty", + "Cross-platform (macOS/Windows/Linux) with a familiar VS Code-derived editor" + ], + "limitations": [ + "Worst security launch record among major agent IDEs: launch-day credential exfiltration via indirect prompt injection (PromptArmor, Rehberger), default auto-execution policies, and webhook.site on the default allowlist", + "Strict Mode sandbox was bypassable to full RCE via -X flag injection in find_by_name (reported 2026-01-07, patched 2026-02-28); no CVE was assigned", + "Google classifies whole prompt-injection exfiltration classes as 'known issues' ineligible for bounty, several inherited unpatched from the Windsurf codebase", + "Consumer interactions feed model training by default (opt-out required) unless governed by Workspace/GCP terms", + "Closed-source pivot: replaced the Apache-2.0 Gemini CLI (shut off for consumers 2026-06-18) with the closed agy binary and ~50x lower free quotas", + "Trojanized-installer malvertising campaign actively targets its download funnel (Malwarebytes, Apr 2026)", + "Preview product: no SLA, no enterprise controls, quotas and terms change without notice" + ], + "metadata": { + "license": "Proprietary (closed-source fork built on non-exclusively licensed Windsurf code; agy CLI is a closed Go binary)", + "supported_models": [ + "Google Gemini 3 Pro / Gemini 3 family (primary)", + "Anthropic Claude Sonnet 4.5/4.6 and Opus 4.6", + "GPT-OSS-120B" + ], + "programming_languages": [ + "All languages supported by the VS Code ecosystem" + ], + "deployment_type": "Local IDE + CLI (macOS/Windows/Linux) with cloud model inference; Google account required", + "tool_support": [ + "Editor, terminal, and embedded browser control", + "Agent manager for parallel agents", + "Artifact generation (plans, task lists, screenshots, recordings)", + "MCP support", + "agy CLI for terminal workflows" + ], + "first_release": "2025-11-18 (public preview, launched with Gemini 3); agy CLI became the consumer path after Gemini CLI's consumer shutdown 2026-06-18", + "pricing": "Free public preview with rate limits (agy CLI free tier ~20 requests/day); post-preview pricing unannounced; enterprise access via Gemini Code Assist licenses", + "security_incidents": "Launch-week data-exfiltration chains via indirect prompt injection (2025-11-25, PromptArmor/Rehberger; several issues inherited from Windsurf, reported to Windsurf May 2025); Strict Mode sandbox escape to RCE via find_by_name -X flag injection (reported 2026-01-07, patched 2026-02-28, disclosed 2026-04-20 by Pillar Security, no CVE); trojanized-installer malvertising campaign (Malwarebytes, Apr 2026)" + }, + "related_entities": [ + "gemini-cli", + "google-jules", + "cursor-agent", + "claude-code" + ], + "tags": [ + "ide", + "agentic-coding", + "parallel-agents", + "proprietary" + ] +} diff --git a/data/agents/goose.json b/data/agents/goose.json new file mode 100644 index 0000000..e8686cc --- /dev/null +++ b/data/agents/goose.json @@ -0,0 +1,474 @@ +{ + "id": "goose", + "type": "agent", + "name": "Goose", + "provider": "Agentic AI Foundation (Linux Foundation); originally Block", + "version": "1.41.0 (2026-07-03)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Open-source, local-first AI agent written in Rust, created by Block and donated to the Linux Foundation's Agentic AI Foundation (repo moved to aaif-goose/goose, April 2026). Ships as a CLI and desktop app, is MCP-native with 70+ extensions, and works with 15+ LLM providers including fully local models via Ollama. Free under Apache-2.0; vendor-neutral foundation governance is its core trust story, with the usual MCP/prompt-injection surface of local agents.", + "website": "https://goose-docs.ai/", + "trust_vector": { + "performance_reliability": { + "overall_score": 77, + "criteria": { + "task_completion_accuracy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Executes coding and workflow tasks end-to-end (editing, running, testing); results depend on the user-selected LLM, from frontier APIs to local Ollama models" + } + ], + "methodology": "Capability review across coding and automation workflows; model-agnostic design means accuracy tracks the configured model", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "MCP-native from the ground up with 70+ extensions and 15+ LLM providers across macOS, Linux, and Windows" + } + ], + "methodology": "Review of MCP extension system maturity and built-in developer tooling reliability", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Autonomous multi-step execution with recipes for repeatable workflows and lead/worker multi-model configurations for planning versus execution" + } + ], + "methodology": "Evaluation of recipes, lead/worker mode, and long-horizon task behavior", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Session persistence and resume, .goosehints project context files, and an optional memory extension; no managed long-term memory service" + } + ], + "methodology": "Review of session storage, hints files, and memory extension capabilities", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Agent loop retries failed commands and self-corrects from tool errors; a repetition inspector guards against runaway loops" + } + ], + "methodology": "Observed recovery behavior from failing commands and the tool-inspection pipeline design", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Subagents and recipe-driven task delegation supported; lead/worker model splits planning and execution across models" + } + ], + "methodology": "Testing of subagent delegation and multi-model orchestration features", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 69, + "criteria": { + "tool_sandboxing": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Goose permission modes documentation", + "url": "https://goose-docs.ai/docs/guides/managing-tools/goose-permissions/", + "date": "2026-05-20", + "value": "No OS-level sandbox by default: goose runs shell and file tools with the user's privileges; permission modes (Auto, Approve, Smart Approve, Chat) gate execution rather than isolate it" + } + ], + "methodology": "Execution isolation review; containerized deployment is possible but user-managed, mitigation is approval-based", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Goose extension allowlist documentation", + "url": "https://block.github.io/goose/docs/guides/allowlist/", + "date": "2026-05-20", + "value": "Administrators can enforce an extension allowlist (GOOSE_ALLOWLIST) restricting which MCP servers can be installed; tool calls pass a stacked inspection pipeline (security, egress, adversary, permission, repetition) before execution" + } + ], + "methodology": "Review of permission modes, allowlist enforcement, and the tool inspection pipeline", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Docs warn goose may follow commands embedded in fetched content; inspectors and approval modes reduce blast radius but the MCP/web-content injection surface matches peer local agents" + } + ], + "methodology": "Injection surface review across MCP extensions and web-fetching tools; defenses are procedural (approvals) rather than architectural", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "Local-first: sessions, keys, and files stay on the user's machine; isolation between projects or extensions is not enforced beyond OS permissions" + } + ], + "methodology": "Data-flow review of local-first architecture and extension access scope", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "Apache-2.0, fully open Rust codebase (50.9K stars), now under vendor-neutral Linux Foundation AAIF governance with open security advisories" + } + ], + "methodology": "License, source availability, and governance review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 85, + "criteria": { + "data_retention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "Local-first design: no vendor cloud stores prompts or code; session data lives on the user's machine" + } + ], + "methodology": "Privacy architecture review of the self-hosted, local-first model", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "As self-hosted software, GDPR posture is determined by the deployer and chosen model provider; fully local operation supports strict data-residency requirements" + } + ], + "methodology": "Compliance capabilities assessment across deployment configurations", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Data flows only to the user-configured LLM provider and any installed MCP extensions; telemetry is limited and controllable" + } + ], + "methodology": "Data-flow analysis across model providers and extensions", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "Runs fully offline with local models via Ollama and other local providers; CLI and desktop are both local applications" + } + ], + "methodology": "Deployment options assessment including air-gapped local-model configurations", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 83, + "criteria": { + "documentation_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Comprehensive docs covering permissions, allowlists, extensions, recipes, and enterprise guidance, maintained under the AAIF" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Full session transcripts of tool calls and outputs; permission decisions persisted in permission.yaml; observability integrations available but not enterprise-grade by default" + } + ], + "methodology": "Review of session logs, permission records, and observability hooks", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Goose permission modes documentation", + "url": "https://goose-docs.ai/docs/guides/managing-tools/goose-permissions/", + "date": "2026-05-20", + "value": "Approval modes surface each intended tool call with allow/deny prompts before execution, making agent intent visible step by step" + } + ], + "methodology": "Assessment of pre-execution visibility and reasoning transparency", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "Apache-2.0 Rust codebase under Linux Foundation AAIF governance, donated by Block alongside Anthropic's MCP and OpenAI's AGENTS.md at the foundation's launch" + }, + { + "source": "Linux Foundation AAIF announcement", + "url": "https://www.linuxfoundation.org/press/linux-foundation-announces-the-formation-of-the-agentic-ai-foundation", + "date": "2025-12-09", + "value": "AAIF formed 2025-12-09 with goose as an anchor project; platinum members include AWS, Anthropic, Block, Bloomberg, Cloudflare, Google, Microsoft, and OpenAI" + } + ], + "methodology": "Open source and governance review; repo migration to aaif-goose completed April 2026", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "50.9K stars, 5.5K forks, active Discord, and steady releases (v1.41.0 on 2026-07-03)" + }, + { + "source": "Goose blog - move to AAIF", + "url": "https://goose-docs.ai/blog/2026/04/07/goose-moves-to-aaif/", + "date": "2026-04-07", + "value": "Repository and docs migrated from block/goose to the Agentic AI Foundation on 2026-04-07, broadening the contributor base beyond Block" + } + ], + "methodology": "Community engagement analysis of stars, release cadence, and post-donation governance", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 77, + "criteria": { + "ease_of_integration": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Single-binary CLI install or desktop app on macOS/Linux/Windows; provider setup is a guided config flow, and MCP extensions install in one step" + } + ], + "methodology": "Setup time and integration surface assessment across CLI and desktop", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "Primarily a single-user local agent; headless/API usage and recipes enable automation, but fleet orchestration is left to the deployer" + } + ], + "methodology": "Assessment of headless usage, automation hooks, and organizational deployment patterns", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Goose GitHub repository (aaif-goose/goose)", + "url": "https://github.com/aaif-goose/goose", + "date": "2026-07-09", + "value": "Free Apache-2.0 software; costs limited to the chosen LLM API, and fully local models make zero-marginal-cost operation possible" + } + ], + "methodology": "Pricing model analysis of free software plus BYO model costs", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Goose documentation", + "url": "https://goose-docs.ai/", + "date": "2026-06-15", + "value": "Session logs and cost/token tracking built in; centralized monitoring, alerting, and audit dashboards require external tooling" + } + ], + "methodology": "Monitoring features assessment for individual and enterprise use", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Goose GitHub repository releases", + "url": "https://github.com/aaif-goose/goose/releases", + "date": "2026-07-09", + "value": "Mature 1.x series (v1.41.0, 2026-07-03) with frequent releases; used internally at Block scale, and foundation governance reduces single-vendor abandonment risk" + } + ], + "methodology": "Maturity assessment from release history, Block production usage, and governance continuity", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 82, + "notes": "Strong local coding agent with LSP-free simplicity and MCP extensibility; quality tracks the chosen model" + }, + "data-analysis": { + "overall": 78, + "notes": "Good for scripted analysis and workflow automation via shell and MCP extensions" + }, + "research-assistant": { + "overall": 74, + "notes": "Capable with web/MCP extensions; injection caution needed when fetching untrusted content" + }, + "content-creation": { + "overall": 70, + "notes": "Serviceable for docs and technical writing; not its design center" + } + }, + "best_for": [ + "Developers wanting a free, open, local-first agent that works with any model including local Ollama", + "Privacy-sensitive teams needing code and prompts to stay on their own machines", + "Organizations that value vendor-neutral foundation governance over single-company tools", + "MCP-heavy workflows leveraging its 70+ extension ecosystem" + ], + "strengths": [ + "Vendor-neutral governance: donated by Block to the Linux Foundation's Agentic AI Foundation (Dec 2025 announcement; repo moved April 2026)", + "Local-first Rust application: fast, no vendor cloud, works fully offline with local models", + "MCP-native with 70+ extensions and 15+ LLM providers", + "Layered controls: permission modes, extension allowlist, and a stacked tool-inspection pipeline", + "Free Apache-2.0 with an active community (50K+ stars) and steady release cadence", + "Both CLI and desktop app across macOS, Linux, and Windows" + ], + "limitations": [ + "No OS-level sandbox by default; shell and file tools run with full user privileges", + "Prompt injection via MCP extensions and fetched web content remains an open risk, mitigated only by approval modes", + "Single-user focus: no built-in multi-tenant, fleet management, or enterprise audit dashboards", + "Output quality varies widely with the configured model, especially small local ones", + "Post-donation transition (repo/docs moves, org rename) creates some ecosystem link rot and tooling churn" + ], + "metadata": { + "license": "Apache-2.0", + "repository": "https://github.com/aaif-goose/goose", + "supported_models": [ + "Anthropic Claude", + "OpenAI GPT models", + "Google Gemini", + "Local models via Ollama", + "15+ providers total" + ], + "languages": [ + "Rust" + ], + "deployment_type": "Local CLI and desktop app (macOS, Linux, Windows); self-hosted", + "architecture": "Local-first Rust agent with MCP-based extension system, permission modes, and recipes", + "first_release": "January 2025 (open-sourced by Block)", + "governance": "Agentic AI Foundation (AAIF) at the Linux Foundation since April 2026 (announced 2025-12-09); founded alongside MCP and AGENTS.md", + "current_version": "1.41.0 (2026-07-03)", + "github_stars": "50900+", + "pricing": "Free; users pay only their chosen LLM provider (or nothing with local models)" + }, + "related": [ + "claude-code", + "gemini-cli", + "cline", + "opencode", + "smolagents" + ], + "tags": [ + "coding-agent", + "open-source", + "local-first", + "mcp", + "foundation-governed" + ] +} diff --git a/data/agents/jetbrains-junie.json b/data/agents/jetbrains-junie.json new file mode 100644 index 0000000..0015211 --- /dev/null +++ b/data/agents/jetbrains-junie.json @@ -0,0 +1,467 @@ +{ + "id": "jetbrains-junie", + "type": "agent", + "name": "Junie", + "provider": "JetBrains", + "version": "GA (out of beta June 2026)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "JetBrains' AI coding agent, out of beta since June 2026 and ranked the #1 coding agent on the independent SWE-Rebench benchmark (61.6% resolved, 72.7% pass@5). Junie plans before it codes, debugs with the IDE's real debugger, reviews pull requests with project context, and runs from JetBrains IDEs, a terminal CLI, or CI (GitHub Actions/GitLab). LLM-agnostic with BYOK and local-model support; included in JetBrains AI subscription tiers.", + "website": "https://junie.jetbrains.com/", + "trust_vector": { + "performance_reliability": { + "overall_score": 80, + "criteria": { + "task_completion_accuracy": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Ranked #1 coding agent on the latest SWE-Rebench run with 61.6% resolved and 72.7% pass@5, ahead of other agents and competitive with raw frontier models" + } + ], + "methodology": "Review of independent SWE-Rebench benchmark placement plus GA feature assessment; SWE-Rebench is a continuously refreshed, contamination-resistant benchmark", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Uses the IDE's real toolchain: debugger (breakpoints, stack inspection, expression evaluation), test runners, and DataGrip-configured database connections for SQL validation" + } + ], + "methodology": "Assessment of IDE-native tool integration (debugger, run configurations, database access) versus shell-only agents", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Plan Mode produces a structured planning document with product requirements, technical design, and delivery stages before generating code" + } + ], + "methodology": "Evaluation of plan-first workflow and plan adherence on multi-file tasks", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Junie by JetBrains", + "url": "https://junie.jetbrains.com/", + "date": "2026-07-09", + "value": "Project guidelines files (.junie/guidelines.md) and IDE project model provide persistent project context; cross-session memory is thinner than dedicated knowledge-base systems" + } + ], + "methodology": "Review of guidelines files, IDE index-backed context, and session persistence behavior", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Agentic debugging sets breakpoints and inspects program state to diagnose failures rather than guess-and-retry with print logging; iterates on failing tests" + } + ], + "methodology": "Assessment of debugger-driven failure diagnosis and test-iteration loops", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Single unified engine across IDE chat, tool window, and CLI via the Agent Communication Protocol (ACP); supports background/long-running tasks and CI-triggered runs, but no native multi-agent orchestration" + } + ], + "methodology": "Review of ACP-based surface unification, background task support, and absence of subagent/fan-out primitives", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 65, + "criteria": { + "tool_sandboxing": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Junie by JetBrains", + "url": "https://junie.jetbrains.com/", + "date": "2026-07-09", + "value": "Runs in the developer's IDE/terminal with action approval modes (including a stricter approval mode versus 'brave' auto-run); no OS-level kernel-enforced sandbox documented" + } + ], + "methodology": "Review of execution model: approval-gated local execution rather than kernel-level isolation; CI runs inherit the pipeline's isolation", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains AI plans and usage", + "url": "https://www.jetbrains.com/help/ai-assistant/licensing-and-subscriptions.html", + "date": "2026-07-09", + "value": "Organization-level administration via JetBrains AI Enterprise and IDE Services: centralized provisioning, quota management, and model allowlisting for teams" + } + ], + "methodology": "Review of enterprise administration, license/quota controls, and per-user authentication (JetBrains Account, API keys, BYOK)", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Junie processes repository content and PR diffs (including via GitHub Actions/GitLab automation); injection-specific mitigations are not publicly documented" + } + ], + "methodology": "Threat surface analysis of PR-review automation and repo content processing; no published injection defense documentation or third-party research found", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains Trust Center", + "url": "https://trust-center.jetbrains.com/", + "date": "2026-07-09", + "value": "SOC 2 Type II report specifically covering Junie is available in the JetBrains Trust Center, alongside penetration testing reports and infrastructure security documentation" + } + ], + "methodology": "Review of Junie-scoped SOC 2 Type II attestation and JetBrains infrastructure security posture", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "JetBrains Junie GitHub repository", + "url": "https://github.com/JetBrains/junie", + "date": "2026-07-09", + "value": "Proprietary product under JetBrains AI Service Terms; public GitHub repo hosts installers, release management, and issue tracking, not agent source code" + } + ], + "methodology": "License and source availability review; open ACP protocol adoption partially offsets the closed core", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 73, + "criteria": { + "data_retention": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains AI plans and usage", + "url": "https://www.jetbrains.com/help/ai-assistant/licensing-and-subscriptions.html", + "date": "2026-07-09", + "value": "JetBrains AI terms commit to not using customer code to train models; detailed collection settings and opt-outs are configurable per organization" + } + ], + "methodology": "Review of JetBrains AI service terms, training commitments, and data collection settings", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains Trust Center", + "url": "https://trust-center.jetbrains.com/", + "date": "2026-07-09", + "value": "SOC 2 Type II (Junie-scoped) and GDPR compliance documentation available; JetBrains is an EU-headquartered (Czech) vendor with long-standing enterprise compliance practice, including ISO 27001-aligned controls per its security knowledge base" + } + ], + "methodology": "Compliance certification and documentation review via the JetBrains Trust Center (NDA-gated reports)", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains Junie GitHub repository", + "url": "https://github.com/JetBrains/junie", + "date": "2026-07-09", + "value": "Default operation routes code to third-party LLM providers (Anthropic, OpenAI, Google); BYOK (including xAI, OpenRouter, GitHub Copilot) and local runtimes (Ollama, LM Studio, LiteLLM) let teams control the data path" + } + ], + "methodology": "Data flow analysis of managed model routing versus BYOK/local alternatives", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Supports any model without lock-in: local runtimes via LiteLLM, LM Studio, and Ollama; JetBrains AI Enterprise offers on-premises deployment for organizations" + } + ], + "methodology": "Deployment options assessment covering local model runtimes and JetBrains AI Enterprise on-prem offering", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 71, + "criteria": { + "documentation_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Junie by JetBrains", + "url": "https://junie.jetbrains.com/", + "date": "2026-07-09", + "value": "Well-maintained documentation across the Junie site, JetBrains Help (licensing, quotas), and GA/CLI blog posts covering IDE, CLI, and CI usage" + } + ], + "methodology": "Documentation completeness and accuracy review across product site, help center, and blogs", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Step-by-step visible actions in the IDE tool window with reviewable diffs; CLI and CI runs produce inspectable logs and PR review comments" + } + ], + "methodology": "Review of in-IDE action visibility, diff review flow, and CI run artifacts", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Plan Mode documents requirements, technical design, and delivery stages before execution; PR reviews explain findings in project context" + } + ], + "methodology": "Assessment of plan document quality and change justification in reviews", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 40, + "confidence": "high", + "evidence": [ + { + "source": "JetBrains Junie GitHub repository", + "url": "https://github.com/JetBrains/junie", + "date": "2026-07-09", + "value": "Closed-source agent; public repo (324 stars) is a distribution and issue-tracking channel only" + } + ], + "methodology": "Open source assessment of agent core and distribution artifacts", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains Junie GitHub repository", + "url": "https://github.com/JetBrains/junie", + "date": "2026-07-09", + "value": "Active Discord community and GitHub issue tracking; backed by JetBrains' large IDE user base, though the standalone Junie community is younger and smaller than incumbent agents'" + } + ], + "methodology": "Community engagement analysis via Discord, GitHub activity, and JetBrains ecosystem reach", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 78, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "Ships inside all JetBrains IDEs with existing project models and run configurations; CLI installs via shell script, Homebrew, or npm; PR review via GitHub Actions and GitLab" + } + ], + "methodology": "Integration surface assessment: zero-setup for the very large existing JetBrains IDE install base", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains blog - Junie CLI beta", + "url": "https://blog.jetbrains.com/junie/2026/03/junie-cli-the-llm-agnostic-coding-agent-is-now-in-beta/", + "date": "2026-03-18", + "value": "CLI and CI/CD execution enable headless automation and PR-review pipelines; primarily a per-developer agent rather than a managed parallel cloud-agent fleet" + } + ], + "methodology": "Assessment of headless/CI usage and parallelism model; no managed cloud sandbox fleet comparable to cloud-agent products", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains AI plans and pricing", + "url": "https://www.jetbrains.com/ai-ides/buy/", + "date": "2026-07-09", + "value": "Fixed subscription tiers: AI Free $0 (small monthly credit quota), AI Pro $10/mo, AI Ultimate $30/mo ($20/mo annual), AI Enterprise custom; credit top-ups at ~$1/credit; BYOK shifts inference cost to the customer's own provider" + } + ], + "methodology": "Pricing model analysis: predictable flat tiers with quota-based credits; complex agentic tasks can exhaust quotas quickly, and quota-to-task mapping is opaque", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "JetBrains AI plans and usage", + "url": "https://www.jetbrains.com/help/ai-assistant/licensing-and-subscriptions.html", + "date": "2026-07-09", + "value": "Per-user credit/quota usage tracking in the IDE and JetBrains Account; enterprise usage administration via JetBrains AI Enterprise, but no agent-specific observability stack (e.g., OTel export)" + } + ], + "methodology": "Review of quota dashboards and enterprise usage administration; limited dedicated agent telemetry", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "JetBrains blog - Junie out of beta", + "url": "https://blog.jetbrains.com/junie/2026/06/junie-coding-agent-out-of-beta/", + "date": "2026-06-17", + "value": "GA June 2026 on a unified ACP-based engine, backed by JetBrains' 25-year track record, millions of IDE users, and Junie-scoped SOC 2 Type II attestation" + } + ], + "methodology": "Maturity assessment combining GA status, vendor stability, and compliance attestation; agent itself has only weeks of GA history", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 85, + "notes": "Top-ranked on SWE-Rebench; strongest inside JetBrains IDEs where it leverages the debugger, inspections, and project model" + }, + "data-analysis": { + "overall": 70, + "notes": "Good for code-centric data work (SQL via DataGrip connections, scripts); not an analytics product" + }, + "research-assistant": { + "overall": 62, + "notes": "Codebase exploration and PR-context research are solid; general research is out of scope" + }, + "education": { + "overall": 72, + "notes": "Plan documents and debugger-driven walkthroughs are instructive; free tier gives students a low-cost entry point" + } + }, + "best_for": [ + "Teams already standardized on JetBrains IDEs wanting a native agent with zero new tooling", + "Developers who value debugger-driven fixing over log-and-retry agent loops", + "Organizations wanting model choice: managed models, BYOK, or fully local runtimes", + "PR review automation in GitHub Actions/GitLab with project-context awareness", + "Cost-sensitive teams preferring flat subscription tiers over usage-metered agents" + ], + "strengths": [ + "Ranked #1 coding agent on independent SWE-Rebench (61.6% resolved, 72.7% pass@5)", + "Debugs with the IDE's real debugger: breakpoints, stack inspection, expression evaluation", + "Plan Mode produces structured requirements/design/delivery-stage documents before coding", + "LLM-agnostic: BYOK (Anthropic, OpenAI, Google, xAI, OpenRouter) and local runtimes (Ollama, LM Studio, LiteLLM)", + "Runs everywhere JetBrains users work: IDE, terminal CLI, GitHub Actions, GitLab", + "Enterprise trust anchored in JetBrains: Junie-scoped SOC 2 Type II, Trust Center, and AI Enterprise on-prem option", + "Predictable flat pricing tiers, including a free tier" + ], + "limitations": [ + "GA only since June 2026; limited production track record and third-party security research", + "Closed-source core; public GitHub repo is distribution-only", + "No OS-level sandbox for local execution; safety relies on approval modes", + "Prompt injection defenses are not publicly documented despite automated PR-review surface", + "Credit quotas are small (roughly 3/10/35 credits per month across Free/Pro/Ultimate) and complex agent tasks consume them quickly", + "No managed cloud agent fleet; parallel/background execution is bounded by local and CI resources", + "Strongest value requires the JetBrains IDE ecosystem; CLI-only users lose the debugger advantage" + ], + "metadata": { + "repository": "https://github.com/JetBrains/junie", + "license": "Proprietary (JetBrains AI Service Terms); public repo for distribution and issues", + "supported_models": [ + "JetBrains-managed frontier models (Anthropic, OpenAI, Google)", + "BYOK: Anthropic, OpenAI, Google, xAI, OpenRouter, GitHub Copilot", + "Local: Ollama, LM Studio, LiteLLM" + ], + "programming_languages": [ + "All languages supported by JetBrains IDEs (Java, Kotlin, Python, Go, JS/TS, PHP, C#, C++, Ruby, Rust, etc.)" + ], + "deployment_type": "IDE-embedded and local CLI; CI via GitHub Actions/GitLab; on-prem via JetBrains AI Enterprise", + "tool_support": [ + "IDE debugger (agentic debugging)", + "Plan Mode", + "PR review (GitHub Actions, GitLab, CLI, plugin)", + "Database access via DataGrip connections", + "Agent Communication Protocol (ACP)" + ], + "first_release": "Preview January 2025; CLI beta March 2026; GA June 2026", + "pricing": "Included in JetBrains AI tiers: AI Free $0, AI Pro $10/mo, AI Ultimate $30/mo ($20/mo annual), AI Enterprise custom; credit-quota based with ~$1/credit top-ups; BYOK available" + }, + "related_entities": [ + "claude-code", + "openai-codex", + "github-copilot-coding-agent", + "cursor-agent", + "gemini-cli" + ], + "tags": [ + "coding-agent", + "ide-native", + "jetbrains", + "llm-agnostic", + "byok" + ] +} diff --git a/data/agents/lovable.json b/data/agents/lovable.json new file mode 100644 index 0000000..bf02ae3 --- /dev/null +++ b/data/agents/lovable.json @@ -0,0 +1,471 @@ +{ + "id": "lovable", + "type": "agent", + "name": "Lovable", + "provider": "Lovable AB", + "version": "Lovable Agent (2026)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Leading prompt-to-app 'vibe coding' platform from Stockholm-based Lovable AB (ex GPT Engineer): natural language in, deployed full-stack web app out, with Supabase-backed Lovable Cloud. Launched Nov 2024; crossed $400M ARR in Feb 2026 with ~8M users ($330M Series B at $6.6B, Dec 2025). The epicenter of the vibe-coding security debate: its own posture is certified (SOC 2, ISO 27001), but apps it generates have repeatedly shipped without proper access controls.", + "website": "https://lovable.dev/", + "trust_vector": { + "performance_reliability": { + "overall_score": 67, + "criteria": { + "task_completion_accuracy": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Lovable added $100M in revenue in a month", + "url": "https://techcrunch.com/2026/03/11/lovable-says-it-added-100m-in-revenue-last-month-alone-with-just-146-employees/", + "date": "2026-03-11", + "value": "8M+ users shipping working apps at scale (25M+ projects, 1M+ new projects weekly by mid-2026) evidences strong completion on prototype- and MVP-class builds" + } + ], + "methodology": "Assessment of prompt-to-working-app completion from adoption data and user reports; reliability drops on complex multi-service apps and long iterative sessions", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Blog - Introducing Lovable Cloud and AI", + "url": "https://lovable.dev/blog/lovable-cloud", + "date": "2025-09-29", + "value": "Lovable Cloud provisions Supabase-backed databases, auth, storage, and edge functions automatically; Lovable AI routes model calls without user API keys" + } + ], + "methodology": "Review of platform tool reliability across code generation, backend provisioning, and deployment", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Documentation", + "url": "https://docs.lovable.dev/", + "date": "2026-06-01", + "value": "Agent mode decomposes feature requests into multi-step build plans across frontend, backend, and integrations, though planning is shallower than spec-first tools" + } + ], + "methodology": "Evaluation of task decomposition and plan adherence on multi-feature builds", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Blog - How Lovable protects your apps automatically", + "url": "https://lovable.dev/blog/how-lovable-protects-your-apps-automatically", + "date": "2026-06-01", + "value": "Project chat history and a 'security memory' of past findings persist per project; knowledge files carry conventions across sessions" + } + ], + "methodology": "Review of project-scoped context, chat history, and knowledge persistence", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Documentation", + "url": "https://docs.lovable.dev/", + "date": "2026-06-01", + "value": "'Try to Fix' loops resolve many build errors automatically, but users commonly report credit-consuming fix loops on stubborn errors" + } + ], + "methodology": "Assessment of automatic error resolution behavior and documented failure loops from user reports", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 58, + "confidence": "low", + "evidence": [ + { + "source": "Lovable Documentation", + "url": "https://docs.lovable.dev/", + "date": "2026-06-01", + "value": "Single-agent model per project with team collaboration features; no parallel multi-agent orchestration comparable to IDE-native agent platforms" + } + ], + "methodology": "Review of multi-agent and team collaboration capabilities", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 51, + "criteria": { + "tool_sandboxing": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Security", + "url": "https://lovable.dev/security", + "date": "2026-06-01", + "value": "Generation and hosting run entirely in Lovable's managed cloud (no local execution); Lovable Cloud is fronted by WAF controls, network isolation, encrypted storage, and adaptive rate limiting" + } + ], + "methodology": "Security architecture review of the managed-cloud execution model; no user-machine exposure, but platform-side controls are the single boundary", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Pricing", + "url": "https://lovable.dev/pricing", + "date": "2026-07-09", + "value": "SSO and security-center features are gated to Business ($50/mo) and Enterprise tiers; SOC 2 Type II and ISO 27001:2022 certifications cover the platform's own controls" + } + ], + "methodology": "Review of identity and org-level controls across plan tiers", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 46, + "confidence": "low", + "evidence": [ + { + "source": "The Hacker News - Guardio VibeScamming benchmark", + "url": "https://thehackernews.com/2025/04/lovable-ai-found-most-vulnerable-to.html", + "date": "2025-04-09", + "value": "Guardio's VibeScamming benchmark scored Lovable 1.8/10 (worst of tested tools): it generated and auto-hosted working phishing pages with minimal jailbreaking, evidencing weak misuse and injection resistance" + } + ], + "methodology": "Assessment of resistance to adversarial prompting and misuse, anchored on the published Guardio benchmark and Proofpoint reports of tens of thousands of Lovable-hosted phishing URLs monthly since Feb 2025", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "The Register - Lovable denies data leak, cites 'intentional behavior'", + "url": "https://www.theregister.com/2026/04/20/lovable_denies_data_leak/", + "date": "2026-04-20", + "value": "A February 2026 backend regression exposed public-project chat histories, source code, and embedded credentials via a BOLA flaw until 2026-04-20; the researcher's report was initially dismissed as 'intentional behavior' before a fix shipped within hours of public disclosure" + }, + { + "source": "NVD - CVE-2025-48757", + "url": "https://nvd.nist.gov/vuln/detail/CVE-2025-48757", + "date": "2025-05-29", + "value": "CVE-2025-48757: Lovable-generated apps shipped without Supabase Row Level Security; a scan of 1,645 apps found ~10.3% exposing sensitive data to unauthenticated requests. No platform patch, remediated via workarounds and later scanning" + } + ], + "methodology": "Review of platform tenant isolation (one confirmed 2026 exposure incident) plus the systemic isolation failure in generated apps (missing RLS); both directions weigh on this score", + "last_verified": "2026-07-09", + "notes": "Systemic risk: the platform's certified posture does not transfer to the apps it produces. The Moltbook breach (1.5M agent API tokens via a client-exposed Supabase key without RLS, Wiz, Feb 2026) was not attributed to Lovable but is the same vibe-coding/no-RLS pattern at ecosystem scale." + }, + "open_source_transparency": { + "score": 32, + "confidence": "high", + "evidence": [ + { + "source": "Lovable", + "url": "https://lovable.dev/", + "date": "2026-06-01", + "value": "Platform and agent are proprietary (the predecessor GPT Engineer was open source); generated app code is exportable to GitHub, which aids auditability of outputs but not of the platform" + } + ], + "methodology": "Source availability assessment; credit for full code export, none for the closed platform", + "last_verified": "2026-07-09" + } + }, + "notes": "Score the product two ways: Lovable's own corporate posture (SOC 2 Type II, ISO 27001, WAF, security scanning) is solid mid-tier; the systemic security of what it produces is the industry's cautionary tale (CVE-2025-48757, ~10% of scanned apps leaking data, phishing abuse at scale)." + }, + "privacy_compliance": { + "overall_score": 54, + "criteria": { + "data_retention": { + "score": 56, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Privacy Policy", + "url": "https://lovable.dev/privacy", + "date": "2026-06-01", + "value": "GDPR-aligned privacy policy and DPA; the Feb-Apr 2026 exposure of public-project chat histories showed prompts/chat data are retained and were insufficiently protected until all public projects were converted private" + } + ], + "methodology": "Review of retention practices and the 2026 incident's implications for stored prompt/chat data", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "Lovable Security", + "url": "https://lovable.dev/security", + "date": "2026-06-01", + "value": "SOC 2 Type I and Type II, ISO 27001:2022 certified (Aug 2025), GDPR DPA available, public trust center at trust.lovable.dev; EU (Swedish) company subject to GDPR natively" + } + ], + "methodology": "Compliance documentation and certification assessment", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Google Cloud Press Corner - Lovable expands collaboration", + "url": "https://www.googlecloudpresscorner.com/2026-06-03-Lovable-Expands-Collaboration-With-Google-Cloud-to-Scale-AI-Powered-Software-Creation", + "date": "2026-06-03", + "value": "Prompts and code are processed by Gemini (default via Lovable AI) and Anthropic Claude via Google Cloud/Vertex under Lovable's multi-year Google Cloud deal; app backends run on Supabase" + } + ], + "methodology": "Data flow analysis across model providers and backend subprocessors", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "Lovable Documentation", + "url": "https://docs.lovable.dev/", + "date": "2026-06-01", + "value": "Fully hosted SaaS; no self-hosted or offline option. Code export to GitHub allows leaving the platform but not running it locally" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 62, + "criteria": { + "documentation_quality": { + "score": 74, + "confidence": "high", + "evidence": [ + { + "source": "Lovable Documentation", + "url": "https://docs.lovable.dev/", + "date": "2026-06-01", + "value": "Well-organized docs covering features, security view, integrations, and a public changelog; security guidance for builders has expanded materially since 2025" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Documentation - Project security view", + "url": "https://docs.lovable.dev/features/security-view", + "date": "2026-06-01", + "value": "Chat history records agent actions per project, all generated code is inspectable/exportable, and the security view surfaces scan findings; there is no step-level tool-call audit trail comparable to agent IDEs" + } + ], + "methodology": "Review of action visibility, code inspectability, and audit trail depth", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Blog - How Lovable protects your apps automatically", + "url": "https://lovable.dev/blog/how-lovable-protects-your-apps-automatically", + "date": "2026-06-01", + "value": "The agent narrates what it builds and the June 2026 security features explain findings and fixes, but architectural decisions are largely implicit for the non-technical target audience" + } + ], + "methodology": "Assessment of build rationale quality, considering the platform's non-developer audience", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "TechCrunch - Lovable becomes a unicorn", + "url": "https://techcrunch.com/2025/07/17/lovable-becomes-a-unicorn-with-200m-series-a-just-8-months-after-launch/", + "date": "2025-07-17", + "value": "Grew out of the open-source GPT Engineer project (50K+ GitHub stars), but the Lovable platform itself is closed source; users own and can export generated code" + } + ], + "methodology": "Open source assessment crediting OSS heritage and code ownership, not platform openness", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "TechCrunch - Lovable added $100M in revenue in a month", + "url": "https://techcrunch.com/2026/03/11/lovable-says-it-added-100m-in-revenue-last-month-alone-with-just-146-employees/", + "date": "2026-03-11", + "value": "~8M users, $400M ARR (Feb 2026), 25M+ projects, and one of the largest builder communities in AI; fast feature cadence through 2026" + } + ], + "methodology": "Community engagement and release cadence analysis", + "last_verified": "2026-07-09" + } + }, + "notes": "Transparency caveat: the April 2026 disclosure was mishandled at first (researcher's report dismissed as 'intentional behavior', HackerOne triage failure) before Lovable published a same-week postmortem admitting its response 'missed the mark' (lovable.dev/blog/our-response-to-the-april-2026-incident, 2026-04-22)." + }, + "operational_excellence": { + "overall_score": 70, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Lovable", + "url": "https://lovable.dev/", + "date": "2026-06-01", + "value": "Prompt-to-published-app in minutes with built-in hosting, auth, and database; the lowest barrier to entry of any evaluated agent, aimed at the 99% who cannot code" + } + ], + "methodology": "Onboarding and integration friction assessment", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Google Cloud Press Corner - Lovable expands collaboration", + "url": "https://www.googlecloudpresscorner.com/2026-06-03-Lovable-Expands-Collaboration-With-Google-Cloud-to-Scale-AI-Powered-Software-Creation", + "date": "2026-06-03", + "value": "Multi-year Google Cloud deal with a 5x infrastructure expansion; platform processes 1M+ new projects weekly and 600M monthly visits" + } + ], + "methodology": "Platform scalability assessment; generated apps inherit Supabase scaling characteristics", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 64, + "confidence": "high", + "evidence": [ + { + "source": "Lovable Pricing", + "url": "https://lovable.dev/pricing", + "date": "2026-07-09", + "value": "Free (5 daily credits, capped ~30/mo); Pro $25/mo for 100 credits; Business $50/mo; Enterprise custom. Credits per message are transparent, but fix-loops burn credits and a separate Lovable Cloud usage layer bills as apps scale" + } + ], + "methodology": "Pricing model analysis; simple tiers offset by variable credit consumption and a second usage-based cloud billing layer", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Lovable Blog - How Lovable protects your apps automatically", + "url": "https://lovable.dev/blog/how-lovable-protects-your-apps-automatically", + "date": "2026-06-01", + "value": "Automatic pre-publish security scan (10-15s), opt-in Deep Security Scan (2-4 min) with Auto-Fix, dependency checks, and a per-project security view; app-level operational observability remains thin" + } + ], + "methodology": "Monitoring and security posture visibility assessment", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Vibe-coding startup Lovable raises $330M at a $6.6B valuation", + "url": "https://techcrunch.com/2025/12/18/vibe-coding-startup-lovable-raises-330m-at-a-6-6b-valuation/", + "date": "2025-12-18", + "value": "Strong vendor viability ($330M Series B at $6.6B closed 2025-12-18; $12B round in talks per Forbes 2026-06-05), but only ~$20M of ARR is enterprise, and generated apps routinely need security hardening before production use" + } + ], + "methodology": "Vendor maturity versus output production-readiness assessment; company trajectory is excellent, generated-app readiness is the persistent gap", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 74, + "notes": "Category-defining for prompt-to-app prototypes and MVPs; not designed for existing codebases, and outputs need security review before production" + }, + "content-creation": { + "overall": 76, + "notes": "Excellent for landing pages, marketing sites, and interactive content shipped directly to hosting" + }, + "education": { + "overall": 72, + "notes": "Free tier and instant results make it a popular way for non-programmers to learn product building, though it teaches little about the underlying code" + }, + "data-analysis": { + "overall": 58, + "notes": "Can build dashboards backed by Supabase, but has no analytics tooling of its own and struggles with complex data workloads" + } + }, + "best_for": [ + "Founders and non-technical builders shipping MVPs, internal tools, and landing pages from prompts", + "Rapid prototyping and idea validation before committing engineering resources", + "Small teams wanting integrated hosting, auth, and database without DevOps", + "Design-forward web apps where speed to publish matters more than architectural control" + ], + "strengths": [ + "Fastest prompt-to-published-app experience in the category; 8M+ users and 25M+ projects validate the workflow", + "Lovable Cloud integrates Supabase database, auth, storage, and model access with zero setup", + "Certified corporate posture: SOC 2 Type I/II, ISO 27001:2022, GDPR DPA, public trust center", + "Meaningful security response since 2025: automatic pre-publish scans, Deep Security Scan with Auto-Fix, security memory, and dependency checks (June 2026)", + "Full code ownership and GitHub export prevent platform lock-in", + "Exceptional commercial trajectory: $400M ARR (Feb 2026) with 146 employees, $6.6B valuation (Dec 2025)" + ], + "limitations": [ + "Systemic output-security problem: CVE-2025-48757 (missing Supabase RLS) exposed ~10.3% of 1,645 scanned apps, and the pattern persists across the vibe-coding ecosystem (cf. the unattributed Moltbook breach exposing 1.5M tokens)", + "Feb-Apr 2026 platform incident exposed public-project chat histories and source code; initial dismissal of the researcher's report damaged disclosure credibility", + "Guardio's VibeScamming benchmark (1.8/10) and Proofpoint data show the platform is heavily abused to generate and host phishing pages", + "Non-technical users cannot evaluate the security of what they ship; scanning helps but is not enforced remediation", + "Credit consumption is unpredictable on error-fix loops, plus a second usage-based cloud billing layer", + "Cloud-only, closed-source platform with limited enterprise controls below the $50/mo Business tier" + ], + "metadata": { + "license": "Proprietary (generated code owned by users, exportable to GitHub)", + "supported_models": [ + "Google Gemini (default via Lovable AI, no user API keys)", + "Anthropic Claude via Google Cloud/Vertex partnership" + ], + "programming_languages": [ + "TypeScript/React frontends with Supabase (Postgres) backends" + ], + "deployment_type": "Fully hosted SaaS (Lovable Cloud on Google Cloud + Supabase); one-click publish with custom domains", + "tool_support": [ + "Supabase database, auth, and storage provisioning", + "Built-in hosting and publishing", + "GitHub two-way sync and code export", + "Security scanning (pre-publish + Deep Security Scan with Auto-Fix)", + "Figma import and visual edits" + ], + "first_release": "November 2024 (evolved from the open-source GPT Engineer project)", + "pricing": "Free (5 daily credits); Pro $25/mo (100 credits); Business $50/mo (SSO, security center); Enterprise custom; separate usage-based Lovable Cloud billing", + "company": "Lovable AB (Stockholm; $200M Series A at $1.8B Jul 2025; $330M Series B at $6.6B closed 2025-12-18 led by CapitalG and Menlo; $400M ARR Feb 2026 with ~8M users and 146 employees; ~$12B round in talks as of Jun 2026)", + "security_incidents": "CVE-2025-48757 (generated apps missing Supabase RLS; disclosed 2025-05-29, no platform patch, mitigated via scanning/workarounds; CVSS scored 8.26-9.3 depending on source); Feb 3 - Apr 20, 2026 backend regression exposing public-project chat histories and source code (fixed within ~2h of public disclosure, postmortem 2026-04-22)" + }, + "related_entities": [ + "replit-agent", + "cursor-agent", + "github-copilot-coding-agent", + "devin" + ], + "tags": [ + "app-builder", + "vibe-coding", + "no-code", + "proprietary" + ] +} diff --git a/data/agents/microsoft-scout.json b/data/agents/microsoft-scout.json new file mode 100644 index 0000000..2dab3c0 --- /dev/null +++ b/data/agents/microsoft-scout.json @@ -0,0 +1,490 @@ +{ + "id": "microsoft-scout", + "type": "agent", + "name": "Microsoft Scout", + "provider": "Microsoft", + "version": "Frontier preview (experimental, announced Build 2026)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Always-on autonomous personal agent for Microsoft 365, built on the open-source OpenClaw runtime. Announced at Build 2026 (2026-06-02) as an experimental Frontier-program release: a Windows/macOS desktop app that reads/writes files, runs shell commands, drives a browser, and manages email, calendar, and Teams. Pairs 2026's strongest enterprise governance — per-agent Entra identity, task-scoped credentials, in-line Purview DLP, human sign-off — with a heavily CVE-burdened OSS core.", + "website": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "trust_vector": { + "performance_reliability": { + "overall_score": 75, + "criteria": { + "task_completion_accuracy": { + "score": 72, + "confidence": "low", + "evidence": [ + { + "source": "Microsoft 365 Blog - Introducing Microsoft Scout", + "url": "https://www.microsoft.com/en-us/microsoft-365/blog/2026/06/02/introducing-microsoft-scout-your-always-on-personal-agent/", + "date": "2026-06-02", + "value": "Proactively schedules meetings across time zones, generates prep materials, blocks calendar time for deliverables, and flags stalled decisions; Microsoft employees dogfooded an early desktop build, but no independent completion data exists yet" + } + ], + "methodology": "Assessment based on vendor-described capabilities and internal dogfooding; experimental preview with no independent benchmark or field data", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Broad, mature tool integrations: file read/write, shell with tiered permissions, Playwright browser automation, Microsoft Graph (mail, calendar, Teams, OneDrive), and bundled Office document skills" + } + ], + "methodology": "Review of the documented tool stack combining OpenClaw's proven runtime tools with first-party Microsoft 365 integrations", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Works through multi-step tasks showing progress in real time; autonomous modes include Heartbeat (15-120 min background check-ins) and Automations (scheduled or condition-triggered tasks)" + } + ], + "methodology": "Evaluation of documented autonomous execution modes and multi-step task handling", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Remembers preferences and decisions across conversations; always-on identity persists as a continuous agent rather than per-session instances" + } + ], + "methodology": "Review of cross-conversation memory and persistent agent identity design", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 68, + "confidence": "low", + "evidence": [ + { + "source": "InfoQ - Microsoft Scout announced at Build 2026", + "url": "https://www.infoq.com/news/2026/06/microsoft-scout-openclaw-build/", + "date": "2026-06-05", + "value": "Inherits OpenClaw's agent loop retry behavior; a policy conformance system continuously checks operation against guidelines, but recovery behavior in unattended always-on operation is unproven at preview stage" + } + ], + "methodology": "Inference from the OpenClaw runtime plus the conformance checking design; no independent reliability data for the experimental release", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Launches specialized sub-agents that run in parallel for research, code review, and complex tasks, reporting results when finished; custom skills via SKILL.md files" + } + ], + "methodology": "Review of sub-agent delegation and skill extensibility", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 64, + "criteria": { + "tool_sandboxing": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Runs directly on the user's Windows/macOS desktop with permissioned access to the file system, shell, and browser — a tiered permission system and sensitive-directory flags, but no VM or OS-level isolation of execution by default" + }, + { + "source": "GitHub - jgamblin/OpenClawCVEs tracker", + "url": "https://github.com/jgamblin/OpenClawCVEs/", + "date": "2026-04-04", + "value": "The underlying OpenClaw runtime logged 137 security advisories between 2026-02-02 and 2026-04-04 alone (~one every 15 hours); its first formal audit (2026-01-25) found 512 vulnerabilities including 8 critical, and hundreds of CVEs are now tracked" + } + ], + "methodology": "Architecture review: desktop-native execution with permission gating rather than isolation, scored against the OpenClaw core's exceptional CVE volume (including RCE CVE-2026-25253 and CVSS 9.9 privilege escalations CVE-2026-22172/CVE-2026-32922)", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft 365 Blog - Introducing Microsoft Scout", + "url": "https://www.microsoft.com/en-us/microsoft-365/blog/2026/06/02/introducing-microsoft-scout-your-always-on-personal-agent/", + "date": "2026-06-02", + "value": "Each agent operates under its own governed Entra identity (not a shared service account); credentials are scoped to the task at hand and redacted from logs and diagnostics; sensitive actions can require human sign-off before proceeding" + }, + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Granular permissions: enable/disable capability categories (file system, shell, browser, M365), per-command shell auto-approve lists, and sensitive directories that always require explicit approval" + } + ], + "methodology": "Review of identity architecture, credential scoping, approval gates, and permission granularity — the strongest agent access-control story evaluated to date", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 52, + "confidence": "low", + "evidence": [ + { + "source": "Help Net Security - Microsoft Scout opens a new category of always-on Autopilots", + "url": "https://www.helpnetsecurity.com/2026/06/03/microsoft-scout-personal-agent/", + "date": "2026-06-03", + "value": "Always-on operation across email, chats, web pages, and documents is a maximal injection surface; human sign-off on sensitive actions and the policy conformance system mitigate, but unattended Heartbeat/Automation runs execute without a user watching" + } + ], + "methodology": "Threat surface analysis of always-on ingestion of untrusted content combined with desktop shell/browser reach; governance gates credit partially offset the OpenClaw heritage and unattended autonomy", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft 365 Blog - Introducing Microsoft Scout", + "url": "https://www.microsoft.com/en-us/microsoft-365/blog/2026/06/02/introducing-microsoft-scout-your-always-on-personal-agent/", + "date": "2026-06-02", + "value": "Microsoft Purview policies including sensitivity labels and data loss prevention are enforced in the moment as the agent acts; M365 data stays within the tenant boundary, though the agent itself spans local files, shell, and web" + } + ], + "methodology": "Review of in-line Purview DLP enforcement and tenant boundaries versus the broad local-plus-cloud reach of a single agent identity", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "The New Stack - Microsoft made the agent runtime free", + "url": "https://thenewstack.io/microsoft-scout-openclaw-runtime/", + "date": "2026-06-04", + "value": "Built openly on the OpenClaw open-source runtime, with Microsoft contributing policy conformance upstream; the Scout product layer, M365 integrations, and models remain proprietary" + } + ], + "methodology": "Source availability assessment: inspectable OSS core (a transparency positive that also exposes its CVE record) with a closed proprietary shell", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 74, + "criteria": { + "data_retention": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Operates under Microsoft 365 enterprise data commitments and existing customer-agreement preview terms; tenant data handling follows M365 retention policies, though Frontier preview terms are less settled than GA services" + } + ], + "methodology": "Review of M365 enterprise retention commitments as applied to a preview service", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Blog - Introducing the First Frontier Suite", + "url": "https://blogs.microsoft.com/blog/2026/03/09/introducing-the-first-frontier-suite-built-on-intelligence-trust/", + "date": "2026-03-09", + "value": "Ships inside the Microsoft 365 compliance umbrella (E7 Frontier Suite bundles E5 security/compliance, Entra Suite, Agent 365) with Purview, eDiscovery, and EU Data Boundary applying to agent activity" + } + ], + "methodology": "Compliance posture assessment under Microsoft's certifications and the E7 suite's integrated compliance stack", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Processing stays within Microsoft's cloud and the local device; no third-party model providers, though browser automation can reach arbitrary sites the user permits" + } + ], + "methodology": "Data flow analysis of first-party processing versus user-permitted web reach", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Desktop application executes locally on Windows 11/macOS 12+ against local files and shell, but requires the M365 cloud, Entra, Frontier enrollment, and cloud model inference — no self-hosted or air-gapped mode" + } + ], + "methodology": "Deployment options assessment: substantial local execution surface tethered to Microsoft's cloud", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 74, + "criteria": { + "documentation_quality": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout documentation", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Full Microsoft Learn documentation set at announcement covering architecture, permissions, autonomous modes, skills, and setup — unusual completeness for an experimental preview" + } + ], + "methodology": "Documentation completeness review of the Learn corpus and admin guidance", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Microsoft launches Scout", + "url": "https://techcrunch.com/2026/06/02/microsoft-launches-scout-an-openclaw-inspired-personal-assistant/", + "date": "2026-06-02", + "value": "Built-in policy conformance system continuously checks operation against set guidelines, with each conformance check producing its own audit trail; real-time progress display and per-agent Entra identity make actions attributable" + } + ], + "methodology": "Review of audit-trail-per-conformance-check design, identity attribution, and live progress visibility", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Shows tool selection and step-by-step progress in real time and asks for approval before sensitive actions (sending email, running commands, writing files)" + } + ], + "methodology": "Assessment of progress narration and pre-action approval clarity; autonomous background runs are less observable in the moment", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "The New Stack - Microsoft made the agent runtime free", + "url": "https://thenewstack.io/microsoft-scout-openclaw-runtime/", + "date": "2026-06-04", + "value": "The OpenClaw runtime core is open source with Microsoft contributing upstream (policy conformance), but Scout's product layer, Graph integrations, and governance plane are closed" + } + ], + "methodology": "Open source assessment of the hybrid OSS-core/proprietary-shell model", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Thurrott - Build 2026: Microsoft unveils Scout", + "url": "https://www.thurrott.com/a-i/336926/build-2026-microsoft-unveils-scout-personal-work-agent-and-new-in-house-ai-models", + "date": "2026-06-02", + "value": "Major Build 2026 launch attention and a very active upstream OpenClaw community, but Scout itself is gated to Frontier participants with a small early-access population" + } + ], + "methodology": "Community engagement analysis of upstream OSS activity versus the gated preview product", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 63, + "criteria": { + "ease_of_integration": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Native M365 integration once running, but access requires Frontier program enrollment, opt-in attestation, Intune configuration, and Copilot licensing — meaningful admin setup before first use" + } + ], + "methodology": "Onboarding friction assessment covering Frontier enrollment, licensing prerequisites, and tenant configuration", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 66, + "confidence": "low", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "One always-on personal agent per user with parallel sub-agents; org-wide fleet management arrives via Agent 365 in E7, but preview capacity limits are undocumented" + } + ], + "methodology": "Assessment of per-user agent model, sub-agent parallelism, and fleet governance via Agent 365", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft Blog - Introducing the First Frontier Suite", + "url": "https://blogs.microsoft.com/blog/2026/03/09/introducing-the-first-frontier-suite-built-on-intelligence-trust/", + "date": "2026-03-09", + "value": "Positioned within the Microsoft 365 E7 Frontier Suite at $99/user/month (GA 2026-05-01), with agent usage billed on top via Copilot Credits" + }, + { + "source": "Microsoft Learn - Usage-based billing with Copilot Credits", + "url": "https://learn.microsoft.com/en-us/microsoft-365/copilot/usage-based-billing-overview-copilot-credits", + "date": "2026-06-16", + "value": "Copilot Credits pay-as-you-go at $0.01/credit went live for agent workloads 2026-06-16; per-task cost varies with model, context retrieved, tool calls, and runtime — an always-on agent makes monthly spend hard to forecast" + } + ], + "methodology": "Pricing analysis of the fixed E7 seat cost stacked with variable usage-based Copilot Credits for a continuously running agent", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Microsoft 365 Blog - Introducing Microsoft Scout", + "url": "https://www.microsoft.com/en-us/microsoft-365/blog/2026/06/02/introducing-microsoft-scout-your-always-on-personal-agent/", + "date": "2026-06-02", + "value": "Per-conformance-check audit trails, Entra identity attribution, Purview enforcement, Intune management, and Copilot Credits cost dashboards give admins end-to-end observability of agent activity and spend" + } + ], + "methodology": "Review of the integrated audit, identity, DLP, device management, and cost monitoring stack", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 48, + "confidence": "high", + "evidence": [ + { + "source": "Microsoft Learn - Microsoft Scout overview", + "url": "https://learn.microsoft.com/en-us/microsoft-scout/overview", + "date": "2026-06-04", + "value": "Explicitly prerelease: experimental Frontier preview with restricted functionality, no GA timeline, and Microsoft's warning that availability and capabilities may change" + } + ], + "methodology": "Maturity assessment: weeks-old experimental preview of a new product category (always-on Autopilots) on a fast-moving OSS core", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "research-assistant": { + "overall": 74, + "notes": "Parallel sub-agent research across M365 content and the web, with results delivered into Office documents" + }, + "data-analysis": { + "overall": 72, + "notes": "Excel skill plus shell access support spreadsheet and scripted analysis within governed boundaries" + }, + "content-creation": { + "overall": 74, + "notes": "Bundled Word/PowerPoint/Loop skills and co-create mode produce and jointly edit documents" + }, + "customer-support": { + "overall": 55, + "notes": "Personal productivity agent, not a customer-facing bot; can triage a user's own inbox but is not built for support queues" + }, + "code-generation": { + "overall": 62, + "notes": "Can edit code, run builds/tests via shell, and delegate code review to sub-agents, but GitHub Copilot remains Microsoft's dedicated coding agent" + } + }, + "best_for": [ + "M365-centric enterprises piloting always-on personal agents under real governance (Entra, Purview, Intune) rather than consumer-grade controls", + "Executives and knowledge workers wanting proactive calendar, email, and deliverable management inside the tenant boundary", + "Security teams that need per-agent identity, task-scoped credentials, and audit trails as hard requirements", + "Organizations already committed to E7/Frontier that want early influence over the Autopilot category" + ], + "not_recommended_for": [ + "Production or business-critical workflows — this is an explicitly experimental preview with no GA timeline", + "Organizations unwilling to accept an agent core (OpenClaw) with hundreds of recent CVEs, even behind Microsoft's governance layer", + "Cost-sensitive teams: $99/user/mo E7 positioning plus unpredictable always-on Copilot Credits burn", + "Tenants outside the Frontier program or without Copilot licensing" + ], + "strengths": [ + "Best-in-class agent governance: each agent gets its own governed Entra identity instead of a shared service account", + "Task-scoped credentials that are redacted from logs and diagnostics", + "In-line Purview enforcement: sensitivity labels and DLP applied at the moment the agent acts", + "Human sign-off gates on sensitive actions, plus granular per-capability and per-command permissions", + "Policy conformance system where every check produces its own audit trail", + "Deep native M365 reach (mail, calendar, Teams, OneDrive, Office skills) with sub-agent delegation and custom SKILL.md skills", + "Open-source OpenClaw core with Microsoft contributing governance work upstream" + ], + "limitations": [ + "Wraps a heavily CVE-burdened OSS core: OpenClaw logged 137 advisories in Feb-Apr 2026 alone, a 512-finding first audit, and critical RCE/privilege-escalation CVEs (CVE-2026-25253, CVE-2026-22172, CVE-2026-32922) — Microsoft's governance layer cannot patch faster than upstream", + "Runs natively on the user's desktop with file/shell/browser reach — permission-gated but not VM-isolated by default", + "Always-on ingestion of email, chats, and web content is a maximal prompt-injection surface, and Heartbeat/Automation runs execute unattended", + "Experimental Frontier preview: restricted functionality, may never reach GA, capabilities can change without notice", + "Expensive and hard to forecast: E7 Frontier Suite at $99/user/mo plus usage-based Copilot Credits ($0.01/credit, live 2026-06-16) for a continuously running agent", + "Heavy prerequisites: Frontier enrollment, opt-in attestation, Intune configuration, and Copilot licensing", + "Windows 11/macOS 12+ desktop only; no web, mobile, or self-hosted option" + ], + "metadata": { + "license": "Proprietary product layer on the open-source OpenClaw runtime (Microsoft contributes policy conformance upstream)", + "supported_models": [ + "Microsoft-hosted models (Copilot stack, including new in-house models announced at Build 2026)" + ], + "deployment_type": "Desktop app (Windows 11+, macOS 12+) with local file/shell/browser execution tied to the Microsoft 365 cloud", + "architecture": "OpenClaw open-source agent runtime + Microsoft governance plane (Entra Agent ID, Purview DLP, Intune, policy conformance system)", + "tool_support": [ + "File system read/write (permissioned)", + "Shell with tiered permissions", + "Playwright browser automation", + "Microsoft 365: mail, calendar, Teams, OneDrive, meetings", + "Bundled Word/Excel/PowerPoint/Loop skills; custom SKILL.md skills", + "Sub-agents, Heartbeat background runs, Automations" + ], + "first_release": "Announced Build 2026 (2026-06-02); experimental preview via the Frontier program", + "pricing": "Positioned with Microsoft 365 E7 Frontier Suite ($99/user/mo, GA 2026-05-01); agent usage billed via Copilot Credits ($0.01/credit pay-as-you-go, live 2026-06-16)" + }, + "related_entities": [ + "openclaw", + "chatgpt-agent", + "claude-cowork", + "github-copilot-coding-agent", + "microsoft-agent-framework" + ], + "tags": [ + "autonomous", + "always-on", + "enterprise", + "microsoft", + "openclaw-based" + ] +} diff --git a/data/agents/openclaw.json b/data/agents/openclaw.json new file mode 100644 index 0000000..be24388 --- /dev/null +++ b/data/agents/openclaw.json @@ -0,0 +1,524 @@ +{ + "id": "openclaw", + "type": "agent", + "name": "OpenClaw", + "provider": "OpenClaw Foundation", + "version": "2026.x (rolling; CVE-2026-28466 patched in 2026.2.14)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Viral MIT-licensed open-source personal AI agent (formerly Clawdbot, then Moltbot) stewarded by the OpenClaw Foundation with OpenAI backing. Runs on the user's own machine and acts through WhatsApp, Telegram, Signal, Discord, and iMessage; it browses, emails, shops, and controls the desktop with any BYO LLM. The fastest repo to ~350K+ GitHub stars and 2026's defining agent-security story: hundreds of CVEs, mass-exposed instances, and malicious ClawHub skills.", + "website": "https://github.com/openclaw/openclaw", + "trust_vector": { + "performance_reliability": { + "overall_score": 75, + "criteria": { + "task_completion_accuracy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "DataCamp - What Is OpenClaw?", + "url": "https://www.datacamp.com/blog/what-is-openclaw", + "date": "2026-06-15", + "value": "Completes real-world personal-assistant tasks (research, email, shopping, scheduling, device control) through messaging apps; quality depends heavily on the BYO LLM configured" + }, + { + "source": "Wikipedia - OpenClaw", + "url": "https://en.wikipedia.org/wiki/OpenClaw", + "date": "2026-07-09", + "value": "Widely adopted general-purpose personal agent, but documented incidents of unprompted autonomous actions (e.g., the February 2026 MoltMatch dating-profile incident)" + } + ], + "methodology": "Review of documented capabilities and incident reports; accuracy varies with the user-configured model and skills", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Gateway/node architecture reliably drives messaging platforms (WhatsApp, Telegram, Signal, Discord, iMessage), browser, email, and desktop control across a large skill ecosystem" + } + ], + "methodology": "Architecture review of gateway-to-node invocation model and breadth of working integrations reported by the community", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "DataCamp - What Is OpenClaw?", + "url": "https://www.datacamp.com/blog/what-is-openclaw", + "date": "2026-06-15", + "value": "Handles long-horizon tasks including scheduled routines and proactive heartbeat check-ins; planning quality is delegated to the configured LLM" + } + ], + "methodology": "Assessment of long-horizon task orchestration, scheduled jobs, and proactive agent behavior", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Persistent local memory files and per-contact context give the agent durable cross-session memory of users, preferences, and prior conversations" + } + ], + "methodology": "Review of local persistent memory design and cross-session continuity behavior", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia - OpenClaw", + "url": "https://en.wikipedia.org/wiki/OpenClaw", + "date": "2026-07-09", + "value": "Autonomous behavior has gone off-script in documented incidents (agents taking actions users never directed); recovery relies on the user noticing via chat" + } + ], + "methodology": "Incident review; always-on autonomy without strong guardrails means errors can compound before a human intervenes", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Multi-agent routing lets one gateway host multiple agent personas across channels; agent-to-agent messaging is community-driven rather than a hardened framework feature" + } + ], + "methodology": "Review of multi-agent routing and community multi-agent usage patterns", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 37, + "criteria": { + "tool_sandboxing": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "GitHub Advisory GHSA-gv46-4xfq-jv58 (CVE-2026-28466)", + "url": "https://github.com/advisories/GHSA-gv46-4xfq-jv58", + "date": "2026-02-14", + "value": "RCE via node-invoke approval bypass (CVSS 9.4): gateway failed to sanitize RPC parameters, letting attackers inject 'approved: true' to bypass exec approval and run arbitrary shell commands; patched in 2026.2.14" + }, + { + "source": "Kaspersky - OpenClaw found unsafe for use", + "url": "https://me-en.kaspersky.com/blog/openclaw-vulnerabilities-exposed/25233/", + "date": "2026-02-10", + "value": "Runs with full user privileges by default (shell, filesystem, email, messaging); a late-January 2026 audit identified 512 vulnerabilities, eight critical" + } + ], + "methodology": "Review of default execution model (unsandboxed, full user privileges) and sandbox-bypass CVE history; Docker isolation is documented but opt-in", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 22, + "confidence": "high", + "evidence": [ + { + "source": "ARMO - CVE-2026-32922 analysis", + "url": "https://www.armosec.io/blog/cve-2026-32922-openclaw-privilege-escalation-cloud-security/", + "date": "2026-03-29", + "value": "CVE-2026-32922 (CVSS 9.9): token-rotation race condition converts a pairing token into full admin control with RCE; ARMO also measured 135,000+ internet-exposed instances in 82 countries, 63% with no authentication" + }, + { + "source": "OpenClaw CVE trackers (jgamblin/OpenClawCVEs)", + "url": "https://github.com/jgamblin/OpenClawCVEs/", + "date": "2026-07-09", + "value": "CVE-2026-22172 (CVSS 9.9) allowed WebSocket clients to self-declare admin scopes, bypassing gateway authentication entirely" + } + ], + "methodology": "Review of authentication CVEs and internet-exposure measurements; two CVSS 9.9 auth-bypass/privilege-escalation flaws plus majority-unauthenticated public deployments", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 20, + "confidence": "high", + "evidence": [ + { + "source": "Wikipedia - OpenClaw (Cisco research)", + "url": "https://en.wikipedia.org/wiki/OpenClaw", + "date": "2026-07-09", + "value": "Susceptible to prompt injection by design: ingests untrusted content from messaging apps, email, and the web while holding broad permissions; Cisco researchers found third-party skills exfiltrating data without user awareness" + }, + { + "source": "Unit 42 - OpenClaw's skill marketplace and the AI supply chain threat", + "url": "https://unit42.paloaltonetworks.com/openclaw-ai-supply-chain-risk/", + "date": "2026-06-20", + "value": "Unit 42 found malicious ClawHub skills that appeared legitimate, including runtime agentic affiliate injection and agentic front-running - novel injection-style attacks executed through the agent itself" + } + ], + "methodology": "Attack-surface review: untrusted inbound channels plus autonomous tool execution with no robust built-in injection defenses; vendor research confirms in-the-wild exploitation", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 28, + "confidence": "high", + "evidence": [ + { + "source": "Kaspersky - OpenClaw found unsafe for use", + "url": "https://me-en.kaspersky.com/blog/openclaw-vulnerabilities-exposed/25233/", + "date": "2026-02-10", + "value": "Agent aggregates credentials, API keys, message history, email, and calendar access in one local trust domain; compromise of the agent exposes everything it touches" + }, + { + "source": "TechRadar - Malicious OpenClaw skills including macOS infostealers", + "url": "https://www.techradar.com/pro/security/multiple-malicious-openclaw-skills-found-online-including-two-macos-infostealers", + "date": "2026-06-18", + "value": "Malicious ClawHub skills delivered macOS infostealers, demonstrating that third-party skills execute inside the agent's full data context" + } + ], + "methodology": "Data-flow review: single-process access to messaging, email, credentials, and files with skills running in the same context", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "MIT-licensed, fully open source (TypeScript/Swift) with public security advisories and rapid patch releases; foundation-governed since February 2026" + } + ], + "methodology": "License, source availability, and advisory-process review; transparency is high even though the security record is poor", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 56, + "criteria": { + "data_retention": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Self-hosted: no vendor retains user data, but the agent persists message history, memories, and credentials in local plaintext files that hundreds of CVEs and malicious skills have targeted" + } + ], + "methodology": "Review of local data persistence model and its practical exposure given the vulnerability history", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 35, + "confidence": "medium", + "evidence": [ + { + "source": "Kaspersky - OpenClaw found unsafe for use", + "url": "https://me-en.kaspersky.com/blog/openclaw-vulnerabilities-exposed/25233/", + "date": "2026-02-10", + "value": "No compliance program, DPA, or certifications; the agent processes third parties' personal messages and contact data with no consent mechanism, and regulators (China, March 2026) have restricted state use" + } + ], + "methodology": "Compliance posture review; hobbyist-governed foundation project with no formal privacy or compliance framework", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 45, + "confidence": "medium", + "evidence": [ + { + "source": "DataCamp - What Is OpenClaw?", + "url": "https://www.datacamp.com/blog/what-is-openclaw", + "date": "2026-06-15", + "value": "BYO-LLM design sends personal message content to whichever model provider is configured (OpenAI, Anthropic, DeepSeek, or local); third-party skills can add undisclosed data flows" + }, + { + "source": "Unit 42 - OpenClaw's skill marketplace and the AI supply chain threat", + "url": "https://unit42.paloaltonetworks.com/openclaw-ai-supply-chain-risk/", + "date": "2026-06-20", + "value": "Documented ClawHub skills exfiltrating data to third parties without user awareness" + } + ], + "methodology": "Data-flow analysis across model providers and the unvetted skill supply chain", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Runs entirely on user hardware (Mac, Linux, VPS, Raspberry Pi) and supports local models, enabling fully self-hosted operation" + } + ], + "methodology": "Deployment options assessment including fully local model configurations", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 78, + "criteria": { + "documentation_quality": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository and docs", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Extensive setup, gateway, and security-hardening docs exist, but breakneck release churn and repeated renames leave documentation lagging the code" + } + ], + "methodology": "Documentation completeness review against actual configuration surface", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Session logs and chat transcripts record agent actions, but always-on background autonomy (heartbeats, scheduled tasks) can act without a human watching" + } + ], + "methodology": "Review of logging, transcript visibility, and unattended-operation blind spots", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia - OpenClaw", + "url": "https://en.wikipedia.org/wiki/OpenClaw", + "date": "2026-07-09", + "value": "Conversational interface explains actions when asked, but incidents like MoltMatch show the agent taking consequential actions users could not anticipate or trace to an instruction" + } + ], + "methodology": "Explainability assessment of autonomous action decisions versus user intent", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "MIT license, fully open codebase, public advisories, foundation governance independent of any single company (OpenAI is primary financial sponsor)" + } + ], + "methodology": "License and source availability review", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 97, + "confidence": "high", + "evidence": [ + { + "source": "Wikipedia - OpenClaw", + "url": "https://en.wikipedia.org/wiki/OpenClaw", + "date": "2026-03-02", + "value": "247,000 stars and 47,700 forks as of 2026-03-02 - the fastest-growing repository in GitHub history (React needed 10+ years to reach 230K)" + }, + { + "source": "OpenClaw statistics roundups", + "url": "https://www.gradually.ai/en/openclaw-statistics/", + "date": "2026-06-30", + "value": "Secondary trackers report ~355K stars in five months and 378K+ by June 2026; figures conflict across sources, so we cite the range 247K (March, Wikipedia) to ~355-378K (June, unofficial)" + } + ], + "methodology": "Community engagement analysis; star counts conflict between Wikipedia (March) and unofficial June trackers, so both are cited", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 55, + "criteria": { + "ease_of_integration": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Wikipedia - OpenClaw (maintainer statement)", + "url": "https://en.wikipedia.org/wiki/OpenClaw", + "date": "2026-07-09", + "value": "Installation is a short CLI flow, but a maintainer warned: 'if you can't understand how to run a command line, this is far too dangerous of a project for you to use safely'" + } + ], + "methodology": "Setup assessment: quick to install, hard to configure safely (auth, network exposure, skill vetting are on the user)", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Designed as a single-user personal agent per gateway; no multi-tenant, fleet, or enterprise orchestration story" + } + ], + "methodology": "Architecture review for multi-user and organizational deployment", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 75, + "confidence": "high", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Free MIT software; costs are BYO LLM tokens, which always-on heartbeats and long chat histories can inflate unpredictably" + } + ], + "methodology": "Pricing model analysis of free software plus variable model-API spend", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 50, + "confidence": "medium", + "evidence": [ + { + "source": "OpenClaw GitHub repository", + "url": "https://github.com/openclaw/openclaw", + "date": "2026-07-09", + "value": "Local logs and web UI exist, but there is no built-in audit trail, alerting, or telemetry suitable for security monitoring of an agent with this much authority" + } + ], + "methodology": "Monitoring and audit capability review against the risk profile of the deployment", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "flyingpenguin - 433 CVEs analysis", + "url": "https://www.flyingpenguin.com/openclaw-is-cooked-433-cves-patched-by-agents-that-cant-fix-whats-broken/", + "date": "2026-05-15", + "value": "~433 accumulated CVEs within 164 days of first ship (138+ by April 2026 per OpenCVE trackers; 413 published in the 2026-05-06 cvelistV5 snapshot) - totals conflict slightly across trackers but all are in the hundreds" + }, + { + "source": "Wikipedia - OpenClaw (China restriction)", + "url": "https://en.wikipedia.org/wiki/OpenClaw", + "date": "2026-07-09", + "value": "March 2026: China restricted state enterprises and government agencies from running OpenClaw; Palo Alto Networks, Cisco, Tenable, and Kaspersky all published advisories" + }, + { + "source": "OpenClaw security news archive", + "url": "https://github.com/joylarkin/openclaw-security-news", + "date": "2026-07-09", + "value": "Curated archive of global government warnings and vendor advisories about OpenClaw, updated continuously through 2026" + } + ], + "methodology": "Production maturity assessment from CVE velocity, government restrictions, vendor advisories, and release churn", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "research-assistant": { + "overall": 62, + "notes": "Genuinely useful always-on assistant for research and errands via chat, if security hardening is applied" + }, + "content-creation": { + "overall": 60, + "notes": "Drafts messages, emails, and posts across channels; output quality depends on the configured model" + }, + "customer-support": { + "overall": 30, + "notes": "Do not expose to untrusted users: majority of internet-facing instances lack auth and injection defenses are weak" + }, + "financial-analysis": { + "overall": 22, + "notes": "Unsuitable: agent-adjacent crypto theft, infostealer skills, and credential aggregation make financial use hazardous" + } + }, + "best_for": [ + "Technical hobbyists who want a self-hosted personal agent and can harden it themselves", + "Experimentation with messaging-native agent UX on isolated, dedicated hardware", + "Researchers studying agent security, skill supply chains, and prompt injection in the wild" + ], + "not_recommended_for": [ + "Non-technical users (a maintainer called it 'far too dangerous' for them)", + "Any enterprise, government, or regulated environment", + "Machines holding credentials, wallets, or sensitive personal/work data", + "Internet-exposed deployment without authentication and network isolation" + ], + "strengths": [ + "Unprecedented community momentum: fastest-growing GitHub repo ever (247K stars by March 2026, ~355-378K reported by June)", + "Acts where users already are: WhatsApp, Telegram, Signal, Discord, iMessage", + "Fully open source (MIT) with foundation governance and OpenAI financial backing", + "Model-agnostic BYO-LLM design, including fully local models", + "Huge skill ecosystem (ClawHub) covering browsing, email, shopping, and desktop control", + "Self-hosted: no vendor cloud holds user data", + "Rapid patch cadence with public security advisories" + ], + "limitations": [ + "Worst security record of any agent in this registry: 138+ CVEs by April 2026 and ~413-433 by mid-2026 (sources conflict), including CVSS 9.9 auth bypass (CVE-2026-22172), CVSS 9.9 privilege escalation (CVE-2026-32922), and RCE via approval bypass (CVE-2026-28466 / GHSA-gv46-4xfq-jv58)", + "135,000+ internet-exposed instances measured, ~63% with no authentication (ARMO, March 2026)", + "ClawHub supply chain compromised at scale: 1,100-1,400 malicious skills reported (including macOS infostealers) plus typosquat campaigns; roughly 1 in 12 packages carried malicious payloads in one audit", + "No default sandboxing: runs with full user privileges over shell, files, email, and messages", + "Prompt injection through inbound messages, email, and web content is structurally unsolved", + "China restricted state/government use (March 2026); Palo Alto, Cisco, Tenable, and Kaspersky issued advisories", + "No compliance posture (no DPA, certifications, or consent handling for third-party message data)" + ], + "metadata": { + "license": "MIT", + "repository": "https://github.com/openclaw/openclaw", + "supported_models": [ + "Anthropic Claude (BYOK)", + "OpenAI GPT models (BYOK)", + "DeepSeek and other API models", + "Local models" + ], + "languages": [ + "TypeScript", + "Swift" + ], + "deployment_type": "Self-hosted (Mac, Linux, VPS, Raspberry Pi)", + "architecture": "Local gateway + node hosts bridging messaging platforms to a BYO LLM, with a directory-based skills system (ClawHub)", + "first_release": "2025-11-24 (as Clawdbot); renamed Moltbot 2026-01-27, OpenClaw 2026-01-30", + "governance": "OpenClaw Foundation (est. Feb 2026, chaired by creator Peter Steinberger, who joined OpenAI 2026-02-14; OpenAI is primary financial sponsor)", + "pricing": "Free (MIT); users pay their own LLM API costs", + "github_stars": "247K (2026-03-02, Wikipedia) to ~355-378K (June 2026, unofficial trackers; sources conflict)", + "security_track_record": "138+ CVEs by 2026-04; ~413-433 by mid-2026 (jgamblin/OpenClawCVEs, OpenCVE, days-since-openclaw-cve counter); 135K+ exposed instances, 63% no auth (ARMO)" + }, + "related": [ + "goose", + "autogpt", + "manus", + "agentgpt", + "claude-code" + ], + "tags": [ + "personal-agent", + "open-source", + "messaging", + "self-hosted", + "high-risk" + ] +} diff --git a/data/agents/opencode.json b/data/agents/opencode.json new file mode 100644 index 0000000..9c9014c --- /dev/null +++ b/data/agents/opencode.json @@ -0,0 +1,484 @@ +{ + "id": "opencode", + "type": "agent", + "name": "OpenCode", + "provider": "SST / Anomaly Innovations", + "version": "1.x (v1.17.17, 2026-07-09)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "MIT-licensed open-source terminal AI coding agent from the SST team (Anomaly Innovations). TUI-first with a client/server architecture, 75+ model providers, an optional curated Zen gateway, and millions of monthly developers. GitHub made Copilot subscriptions work with it (Jan 2026), while Anthropic blocked its consumer-OAuth access the same month, forcing removal of Claude Pro/Max support - a live case study in agent supply-chain and ToS risk.", + "website": "https://opencode.ai/", + "trust_vector": { + "performance_reliability": { + "overall_score": 78, + "criteria": { + "task_completion_accuracy": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode GitHub repository", + "url": "https://github.com/sst/opencode", + "date": "2026-07-09", + "value": "Completes end-to-end coding tasks in the terminal; adoption at 184K stars and millions of monthly developers signals sustained real-world effectiveness" + } + ], + "methodology": "Adoption-signal and hands-on assessment of terminal coding workflows; accuracy tracks the configured model", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Built-in edit, bash, and search tools with LSP integration for language-aware edits; MCP support extends the toolset" + } + ], + "methodology": "Tool invocation testing across file edits, shell execution, LSP-assisted changes, and MCP servers", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode GitHub repository", + "url": "https://github.com/sst/opencode", + "date": "2026-07-09", + "value": "Ships a read-only 'plan' agent that requests permission before bash execution and a full-access 'build' agent, separating strategy from execution" + } + ], + "methodology": "Evaluation of plan/build agent separation on multi-file tasks", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Persistent sessions with resume and share, AGENTS.md project context files, and per-project config; no managed long-term memory" + } + ], + "methodology": "Review of session persistence, AGENTS.md context, and cross-session continuity", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Agent loop iterates on compiler, test, and LSP diagnostics to self-correct; session history allows reverting" + } + ], + "methodology": "Observed recovery behavior from failing builds and diagnostics during evaluation", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Configurable custom agents and subagent delegation; client/server design lets multiple clients drive the same server" + } + ], + "methodology": "Review of custom agent configuration and delegation features", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 67, + "criteria": { + "tool_sandboxing": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "No OS-level sandbox: bash and edit tools operate directly on the host; permission rules (ask/allow/deny per tool and command pattern) gate execution rather than isolate it" + } + ], + "methodology": "Execution isolation review; containerization is left to the user", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Granular permission config per agent (edit/bash/webfetch ask-allow-deny with glob patterns) and a read-only plan agent; no enterprise policy or SSO layer in the open-source core" + } + ], + "methodology": "Assessment of the permission rule system and agent-level restrictions", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "No dedicated injection detection: untrusted repo content and fetched web pages enter the context, with permission prompts as the main backstop - the standard exposure for terminal coding agents" + } + ], + "methodology": "Injection surface review across webfetch, repository content, and MCP servers", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Local-first: code and prompts flow only to the configured provider; the optional share feature uploads session transcripts to opencode.ai and is disabled by config for sensitive environments" + } + ], + "methodology": "Data-flow review of local operation, share-link uploads, and Zen gateway routing", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "OpenCode GitHub repository", + "url": "https://github.com/sst/opencode", + "date": "2026-07-09", + "value": "MIT license, fully open TypeScript codebase with public issue tracker and very high release velocity" + } + ], + "methodology": "License and source availability review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 80, + "criteria": { + "data_retention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "No vendor retention in default BYOK operation; only opt-in share links and optional Zen usage send data to OpenCode-operated services" + } + ], + "methodology": "Privacy architecture review of default local operation versus opt-in cloud features", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode website", + "url": "https://opencode.ai/", + "date": "2026-06-15", + "value": "Compliance posture inherits from the chosen provider; Anomaly publishes standard policies but no formal certification program for the open-source tool" + } + ], + "methodology": "Compliance capabilities assessment across BYOK and Zen configurations", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode Zen documentation", + "url": "https://opencode.ai/docs/zen/", + "date": "2026-06-15", + "value": "Data goes to the user-selected provider among 75+; the optional Zen gateway routes through curated hosted models on pay-per-use terms" + } + ], + "methodology": "Data-flow analysis across direct provider connections, Zen routing, and share links", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Runs locally with local models (Ollama and OpenAI-compatible endpoints) among its 75+ providers; share and Zen are optional" + } + ], + "methodology": "Deployment options assessment including fully local model configurations", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 85, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Clear docs for config, agents, permissions, providers, MCP, Zen, and the SDK/server API" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Full session transcripts of tool calls in the TUI, shareable session links for review, and a server API exposing session state" + } + ], + "methodology": "Review of transcripts, share links, and server-API introspection", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode GitHub repository", + "url": "https://github.com/sst/opencode", + "date": "2026-07-09", + "value": "Plan agent externalizes intended changes read-only before execution; permission prompts surface each sensitive action" + } + ], + "methodology": "Assessment of plan-mode previews and pre-execution visibility", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenCode GitHub repository", + "url": "https://github.com/sst/opencode", + "date": "2026-07-09", + "value": "MIT-licensed, fully open codebase maintained in public by the SST/Anomaly team" + } + ], + "methodology": "Open source assessment of license and codebase completeness", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 93, + "confidence": "high", + "evidence": [ + { + "source": "OpenCode GitHub repository", + "url": "https://github.com/sst/opencode", + "date": "2026-07-09", + "value": "184K stars and 22.9K forks observed 2026-07-09 (secondary write-ups from earlier in 2026 cite 150-172K - the figure moved fast); releases land near-daily (v1.17.17 on 2026-07-09)" + }, + { + "source": "Tech Funding News - OpenCode background story", + "url": "https://techfundingnews.com/opencode-the-background-story-on-the-most-popular-open-source-coding-agent-in-the-world/", + "date": "2026-05-20", + "value": "Reported ~7.5M monthly active developers by May 2026 (other 2026 write-ups cite ~6.5M; estimates vary by methodology)" + } + ], + "methodology": "Community engagement analysis; star and MAU figures conflict across 2026 sources, so ranges are cited with the GitHub-observed value as primary", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 76, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "One-line install script or npm/brew; runs in any terminal over SSH, with a beta desktop app and GitHub Copilot subscription auth since January 2026" + }, + { + "source": "GitHub Changelog - Copilot supports OpenCode", + "url": "https://github.blog/changelog/2026-01-16-github-copilot-now-supports-opencode/", + "date": "2026-01-16", + "value": "GitHub officially enabled Copilot Pro/Pro+/Business/Enterprise subscriptions as an auth/model path for OpenCode" + } + ], + "methodology": "Setup time and integration surface assessment across terminal, SSH, and subscription auth paths", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Client/server architecture supports remote servers, multiple concurrent sessions, and headless/scripted use for CI" + } + ], + "methodology": "Assessment of the server API, headless operation, and parallel session support", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode Zen", + "url": "https://opencode.ai/zen", + "date": "2026-06-15", + "value": "Free MIT tool; BYOK tokens are pay-as-you-go, Zen is pay-per-use, and Copilot subscriptions offer a capped-cost path since January 2026" + } + ], + "methodology": "Pricing analysis across BYOK, Zen, and subscription auth options", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode documentation", + "url": "https://opencode.ai/docs/", + "date": "2026-06-15", + "value": "Per-session cost/token display and share links for review; no built-in fleet dashboards, audit exports, or alerting" + } + ], + "methodology": "Monitoring features assessment for individual and team usage", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "OpenCode GitHub repository", + "url": "https://github.com/sst/opencode", + "date": "2026-07-09", + "value": "Stable 1.x with near-daily releases and massive adoption, but very fast churn; several-million-dollar Zen revenue funds maintenance" + }, + { + "source": "Hacker News - Anthropic blocks third-party subscription access", + "url": "https://news.ycombinator.com/item?id=46549823", + "date": "2026-01-09", + "value": "On 2026-01-09 Anthropic's server-side checks cut OpenCode users off from Claude Pro/Max OAuth overnight without warning; the ToS ban was formalized 2026-02-19 and fully enforced 2026-04-04, forcing OpenCode to drop consumer Claude login" + }, + { + "source": "The Register - Anthropic clarifies third-party access ban", + "url": "https://www.theregister.com/2026/02/20/anthropic_clarifies_ban_third_party_claude_access/", + "date": "2026-02-20", + "value": "Anthropic's clarified terms bar Free/Pro/Max OAuth tokens in third-party tools, a standing upstream-dependency risk for BYOK agents like OpenCode" + } + ], + "methodology": "Maturity assessment from release history and the demonstrated upstream provider-access risk (Anthropic OAuth removal)", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 85, + "notes": "Excellent terminal-native coding agent with LSP-aware edits and plan/build agent separation" + }, + "data-analysis": { + "overall": 75, + "notes": "Solid for scripted analysis through shell tooling; not a notebook environment" + }, + "research-assistant": { + "overall": 70, + "notes": "Capable codebase and web research; webfetch content should be treated as untrusted" + }, + "content-creation": { + "overall": 64, + "notes": "Usable for technical docs; terminal workflow is not aimed at content work" + } + }, + "best_for": [ + "Terminal-first developers who want an open, provider-neutral alternative to Claude Code", + "GitHub Copilot subscribers wanting agentic coding outside VS Code/JetBrains (official since Jan 2026)", + "Remote/SSH development thanks to the client/server TUI architecture", + "Teams avoiding single-vendor lock-in with 75+ model providers including local models" + ], + "strengths": [ + "One of the most popular open-source coding agents: 184K GitHub stars observed July 2026 (150-172K cited in earlier 2026 write-ups) and millions of monthly developers (~6.5-7.5M reported)", + "MIT-licensed, fully open, with near-daily releases", + "Provider-neutral: 75+ providers plus the optional curated Zen gateway", + "Official GitHub Copilot subscription support (January 2026)", + "Client/server TUI works over SSH; plan agent and granular permission rules gate risky actions", + "Local-first BYOK: code goes only to the provider the user chooses" + ], + "limitations": [ + "Anthropic blocked its consumer OAuth access (Jan 2026; ToS formalized Feb, enforced Apr 2026), removing Claude Pro/Max support overnight - proof that upstream providers can cut off access without notice", + "No sandboxing: permission rules are the only barrier to host shell and filesystem", + "No dedicated prompt-injection defenses beyond permission prompts", + "Opt-in share links upload session transcripts to opencode.ai and need disabling in sensitive environments", + "Very fast release churn can break configs and integrations", + "No enterprise governance layer (SSO, audit, policy) in the open-source core" + ], + "metadata": { + "license": "MIT", + "repository": "https://github.com/sst/opencode", + "package_name": "opencode-ai (npm)", + "supported_models": [ + "75+ providers (Anthropic API, OpenAI, Google, GitHub Copilot, OpenRouter, local Ollama, etc.)", + "OpenCode Zen curated gateway (optional, pay-per-use)" + ], + "languages": [ + "TypeScript" + ], + "deployment_type": "Local terminal TUI with client/server architecture; beta desktop app; self-hosted", + "architecture": "TUI client + local server with built-in edit/bash/webfetch tools, LSP integration, MCP support, and configurable agents/permissions", + "first_release": "2024 (SST team); rose to prominence 2025-2026", + "adoption": "184K stars / 22.9K forks (2026-07-09); ~6.5-7.5M monthly developers reported in 2026 (estimates vary)", + "pricing": "Free (MIT) + BYOK provider costs; optional pay-per-use Zen models; works with GitHub Copilot subscriptions", + "github_stars": "184000 (2026-07-09); earlier 2026 sources cite 150-172K" + }, + "related": [ + "claude-code", + "cline", + "gemini-cli", + "github-copilot-coding-agent", + "openai-codex" + ], + "tags": [ + "coding-agent", + "terminal", + "open-source", + "byok", + "multi-provider" + ] +} diff --git a/data/agents/perplexity-comet.json b/data/agents/perplexity-comet.json new file mode 100644 index 0000000..ab870d1 --- /dev/null +++ b/data/agents/perplexity-comet.json @@ -0,0 +1,472 @@ +{ + "id": "perplexity-comet", + "type": "agent", + "name": "Comet", + "provider": "Perplexity AI", + "version": "1.x", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Perplexity's Chromium-based agentic browser. Comet Assistant runs in a sidebar with full context of open tabs and can fill forms, book travel, buy products, and manage email and calendars on the user's behalf. Launched on desktop 2025-07-09, Android 2025-11-20, and iOS 2026-03-18, with Comet Enterprise arriving March 2026. Comet is the canonical browser-agent risk surface: the 'CometJacking' prompt-injection class (2025) was demonstrated against it.", + "website": "https://www.perplexity.ai/comet", + "trust_vector": { + "performance_reliability": { + "overall_score": 73, + "criteria": { + "task_completion_accuracy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Wikipedia - Comet (browser)", + "url": "https://en.wikipedia.org/wiki/Comet_(browser)", + "date": "2026-06-15", + "value": "Assistant handles summarization, email drafting, and product purchases directly from the browsing context; agentic completion degrades on complex multi-page checkout and booking flows" + } + ], + "methodology": "Assessment of assistant task completion across summarization, form-filling, booking, and purchase flows from vendor documentation and independent reviews", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Seraphic Security - Perplexity Comet Browser Key Features", + "url": "https://seraphicsecurity.com/learn/ai-browser/perplexity-comet-browser-key-features-reviews-and-security-tips/", + "date": "2026-01-20", + "value": "Comet Assistant reliably interacts with web content, tabs, email, and calendars as its native tool surface inside the Chromium engine" + } + ], + "methodology": "Review of the assistant's reliability operating browser-native tools: tab context, DOM interaction, email and calendar connectors", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Perplexity Blog - Comet Enterprise is here", + "url": "https://www.perplexity.ai/hub/blog/comet-enterprise-is-here", + "date": "2026-03-17", + "value": "Comet Assistant performs autonomous multi-step tasks like booking flights, managing email, and filling forms for enterprise users" + } + ], + "methodology": "Evaluation of multi-step agentic task orchestration across pages and services", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Comet Data Privacy & Security FAQs", + "url": "https://www.perplexity.ai/comet/resources/articles/comet-data-privacy-security-faq-s", + "date": "2026-04-01", + "value": "Browsing data, history, and assistant memory are stored on-device by default and persist across sessions; personal context is sent to Perplexity only when a personal search requires it" + } + ], + "methodology": "Review of local memory, history-aware assistance, and cross-session personalization", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 68, + "confidence": "low", + "evidence": [ + { + "source": "Wikipedia - Comet (browser)", + "url": "https://en.wikipedia.org/wiki/Comet_(browser)", + "date": "2026-06-15", + "value": "Assistant retries and asks the user for guidance when page automation fails, but independent reviews report stalls and abandoned tasks on dynamic or login-gated sites" + } + ], + "methodology": "Assessment of failure handling on dynamic sites from independent reviews; no vendor-published recovery metrics", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "Perplexity Changelog - Comet iOS Launch and Computer Updates", + "url": "https://www.perplexity.ai/changelog/what-we-shipped--march-27-2026", + "date": "2026-03-27", + "value": "Single sidebar assistant per browsing context; background task execution exists but there is no user-facing multi-agent composition or delegation" + } + ], + "methodology": "Review of concurrent/background task support versus true multi-agent orchestration", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 49, + "criteria": { + "tool_sandboxing": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "Help Net Security - Comet browser exposed users to system-level attacks", + "url": "https://www.helpnetsecurity.com/2025/11/20/perplexity-comet-browser-security-mcp-api/", + "date": "2025-11-20", + "value": "SquareX found Comet's embedded MCP API could allow perplexity.ai-privileged code to execute commands on the user's device; the agent itself operates inside the user's fully authenticated browser session rather than an isolated sandbox" + } + ], + "methodology": "Security architecture review: the assistant acts with the privileges of the logged-in browser session; Chromium process sandboxing does not contain agent actions", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Perplexity Blog - Comet Enterprise is here", + "url": "https://www.perplexity.ai/hub/blog/comet-enterprise-is-here", + "date": "2026-03-17", + "value": "Comet Enterprise (March 2026) adds central admin dashboard, 500+ Chromium policies, MDM deployment, and CrowdStrike Falcon integration; consumer Comet has per-connector permissions but the assistant otherwise inherits full session access" + } + ], + "methodology": "Review of enterprise admin controls against the consumer default of full-session agent privileges", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "LayerX - CometJacking: How One Click Can Turn Perplexity's Comet AI Browser Against You", + "url": "https://layerxsecurity.com/blog/cometjacking-how-one-click-can-turn-perplexitys-comet-ai-browser-against-you/", + "date": "2025-10-02", + "value": "CometJacking: malicious instructions in a URL's collection parameter steer the assistant to read connected Gmail/Calendar data and exfiltrate it base64-encoded; Perplexity initially rejected the August 2025 reports as having no security impact" + }, + { + "source": "Brave - Agentic Browser Security: Indirect Prompt Injection in Perplexity Comet", + "url": "https://brave.com/blog/comet-prompt-injection/", + "date": "2025-08-20", + "value": "Brave showed Comet fed webpage content to its LLM without distinguishing user instructions from untrusted page content, allowing indirect prompt injection (e.g., hidden Reddit comment instructions) to drive cross-site actions on the user's accounts" + } + ], + "methodology": "Analysis of published prompt-injection research (Brave August 2025, LayerX CometJacking October 2025) and vendor response; patches shipped but the agentic-browser injection class remains structurally unsolved", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Comet Browser Help Center - Browsing Privacy & Safety", + "url": "https://comet-help.perplexity.ai/en/articles/12867356-browsing-privacy-safety", + "date": "2026-04-01", + "value": "Browsing data stored on-device by default; personal data is sent to Perplexity servers only when the user initiates an assistant task that requires page or account context" + } + ], + "methodology": "Review of local-first storage claims against the reality that assistant tasks ship page content and connected-account data to Perplexity's cloud", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Wikipedia - Comet (browser)", + "url": "https://en.wikipedia.org/wiki/Comet_(browser)", + "date": "2026-06-15", + "value": "Built on open-source Chromium/Blink, but the Comet assistant layer, agent harness, and Perplexity backend are proprietary and unauditable" + } + ], + "methodology": "Source availability assessment: open engine, closed agent stack", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 55, + "criteria": { + "data_retention": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Comet Data Privacy & Security FAQs", + "url": "https://www.perplexity.ai/comet/resources/articles/comet-data-privacy-security-faq-s", + "date": "2026-04-01", + "value": "Local-first storage with user-controllable history; assistant queries and page context sent to Perplexity are governed by its consumer privacy policy, with opt-out of training retention" + } + ], + "methodology": "Review of on-device defaults versus cloud retention of assistant interactions", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "Perplexity Blog - Comet Enterprise is here", + "url": "https://www.perplexity.ai/hub/blog/comet-enterprise-is-here", + "date": "2026-03-17", + "value": "Enterprise tier ships with admin governance, SSO, and enterprise data-handling terms building on Perplexity Enterprise's SOC 2 posture; consumer tier relies on standard privacy policy with fewer guarantees" + } + ], + "methodology": "Compliance documentation assessment across consumer and enterprise tiers", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "Seraphic Security - Perplexity Comet Browser Key Features", + "url": "https://seraphicsecurity.com/learn/ai-browser/perplexity-comet-browser-key-features-reviews-and-security-tips/", + "date": "2026-01-20", + "value": "Assistant tasks route page content and connected Gmail/Calendar data through Perplexity, which itself orchestrates multiple frontier model backends, widening the data-processing surface" + } + ], + "methodology": "Data flow analysis of connected-account context routed through Perplexity's multi-model backend", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 40, + "confidence": "high", + "evidence": [ + { + "source": "Comet Browser Help Center - Browsing Privacy & Safety", + "url": "https://comet-help.perplexity.ai/en/articles/12867356-browsing-privacy-safety", + "date": "2026-04-01", + "value": "Browser and its data live locally, but every assistant/agentic feature requires Perplexity cloud inference; there is no offline or self-hosted assistant mode" + } + ], + "methodology": "Deployment options assessment: local client, cloud-only agent", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 62, + "criteria": { + "documentation_quality": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Comet Browser Help Center", + "url": "https://comet-help.perplexity.ai/", + "date": "2026-04-01", + "value": "Help center, privacy/security FAQs, and changelog cover features and data handling; threat-model and injection-hardening documentation is thin relative to the attack surface" + } + ], + "methodology": "Documentation completeness review against the product's risk surface", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Perplexity Changelog - Comet iOS Launch and Computer Updates", + "url": "https://www.perplexity.ai/changelog/what-we-shipped--march-27-2026", + "date": "2026-03-27", + "value": "Assistant narrates steps in the sidebar and shows page actions as they happen; there is no exportable audit log of agent actions for consumers" + } + ], + "methodology": "Review of visible step narration versus durable audit trails", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "Seraphic Security - Perplexity Comet Browser Key Features", + "url": "https://seraphicsecurity.com/learn/ai-browser/perplexity-comet-browser-key-features-reviews-and-security-tips/", + "date": "2026-01-20", + "value": "Assistant explains intended actions conversationally with citations for research answers, but model routing and action-selection internals are opaque" + } + ], + "methodology": "Assessment of action explanation and citation quality", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Wikipedia - Comet (browser)", + "url": "https://en.wikipedia.org/wiki/Comet_(browser)", + "date": "2026-06-15", + "value": "Chromium base is open source; the Comet assistant, agent harness, and connectors are closed" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "TechTimes - Perplexity Raises $200 Million for Comet", + "url": "https://www.techtimes.com/articles/318028/20260608/perplexity-raises-200-million-comet-ai-browser-agent-economy-front-door.htm", + "date": "2026-06-08", + "value": "~3 million monthly active users by Q1 2026, top-3 iOS App Store ranking at the March 2026 launch, and rapid release cadence across four platforms" + } + ], + "methodology": "Adoption, release cadence, and public engagement analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 75, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Wikipedia - Comet (browser)", + "url": "https://en.wikipedia.org/wiki/Comet_(browser)", + "date": "2026-06-15", + "value": "Standard Chromium browser: imports bookmarks, passwords, and extensions; free download since October 2025 on Windows, macOS, Android (2025-11-20), and iOS (2026-03-18)" + } + ], + "methodology": "Onboarding friction assessment for consumers switching browsers", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Perplexity Blog - Comet Enterprise is here", + "url": "https://www.perplexity.ai/hub/blog/comet-enterprise-is-here", + "date": "2026-03-17", + "value": "Silent MDM installers deploy Comet across thousands of managed macOS/Windows devices; assistant throughput is bounded by per-user Perplexity plan limits" + } + ], + "methodology": "Scalability assessment of fleet deployment and per-user agent limits", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "TechTimes - Perplexity Raises $200 Million for Comet", + "url": "https://www.techtimes.com/articles/318028/20260608/perplexity-raises-200-million-comet-ai-browser-agent-economy-front-door.htm", + "date": "2026-06-08", + "value": "Comet free worldwide since 2025-10-02; Comet Plus $5/mo (80% publisher revenue share), Perplexity Pro $20/mo and Max $200/mo unlock heavier assistant usage — flat tiers with no usage metering" + } + ], + "methodology": "Pricing model analysis; flat subscription tiers are highly predictable", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Perplexity Blog - Comet Enterprise is here", + "url": "https://www.perplexity.ai/hub/blog/comet-enterprise-is-here", + "date": "2026-03-17", + "value": "Enterprise dashboard provides rollout and policy management plus CrowdStrike Falcon telemetry; consumer edition has no agent-action monitoring or audit tooling" + } + ], + "methodology": "Monitoring and administration features assessment across tiers", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "TechTimes - Perplexity Raises $200 Million for Comet", + "url": "https://www.techtimes.com/articles/318028/20260608/perplexity-raises-200-million-comet-ai-browser-agent-economy-front-door.htm", + "date": "2026-06-08", + "value": "Perplexity raised ~$200M at a ~$20B valuation (finalized June 2026) largely to scale Comet; all-platform availability and enterprise tier signal maturity, while the 2025 injection research record signals an evolving security posture" + } + ], + "methodology": "Maturity assessment weighing funding, platform coverage, and enterprise adoption against a young and actively exploited agent security surface", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "research-assistant": { + "overall": 84, + "notes": "Strongest use case: Perplexity search plus tab-aware summarization, comparison, and cited answers directly in the browsing flow" + }, + "content-creation": { + "overall": 68, + "notes": "Drafts emails and posts from page context well; long-form authoring is better served by dedicated tools" + }, + "data-analysis": { + "overall": 60, + "notes": "Can extract and compare data across open tabs, but has no code execution or spreadsheet-grade analysis environment" + }, + "education": { + "overall": 72, + "notes": "Cited, in-page explanations make it a capable study companion; free tier keeps it accessible" + } + }, + "best_for": [ + "Individuals who want research, summarization, and drafting embedded in everyday browsing", + "Users delegating low-stakes web chores: form filling, comparison shopping, tab triage, email drafting", + "Perplexity Pro/Max subscribers who want their subscription to extend into an agentic browser", + "Enterprises piloting agentic browsing under MDM policies and CrowdStrike-integrated controls (March 2026 tier)" + ], + "not_recommended_for": [ + "Autonomous operation while logged into banking, healthcare, or other high-value accounts", + "Users who click untrusted links: CometJacking-class URL payloads have weaponized the assistant against connected accounts", + "Organizations that cannot tolerate page content and connected-account data flowing to Perplexity's cloud", + "Workflows requiring durable audit logs of every agent action" + ], + "strengths": [ + "First-mover agentic browser with full platform coverage: desktop 2025-07-09, Android 2025-11-20, iOS 2026-03-18", + "Sidebar assistant with whole-tab context handles forms, bookings, purchases, email, and calendar tasks in place", + "Free since October 2025, with flat $5/$20/$200 tiers — no usage metering", + "Local-first data storage; personal context leaves the device only when an assistant task needs it", + "Comet Enterprise (March 2026): MDM deployment, 500+ Chromium policies, CrowdStrike Falcon integration", + "Well-funded roadmap: ~$200M raise at ~$20B valuation (June 2026) explicitly aimed at scaling Comet" + ], + "limitations": [ + "Canonical prompt-injection target: Brave's indirect-injection findings (Aug 2025) and LayerX's CometJacking (Oct 2025) showed one click could exfiltrate connected Gmail/Calendar data", + "Perplexity initially dismissed the CometJacking reports as having no security impact before hardening", + "Assistant acts with the privileges of the user's fully authenticated browser session — no privilege separation", + "SquareX research (Nov 2025) showed the embedded MCP API could enable device-level command execution", + "Closed-source assistant layer prevents independent audit of injection defenses", + "All agentic features require Perplexity cloud inference; no offline or self-hosted mode", + "No consumer-facing audit trail of actions the agent has taken" + ], + "metadata": { + "license": "Proprietary (Chromium base is open source)", + "supported_models": [ + "Perplexity Sonar", + "Frontier models via Perplexity backend (GPT, Claude, Gemini routing)" + ], + "architecture": "Chromium/Blink browser with cloud-backed sidebar agent (Comet Assistant) holding tab context and account connectors", + "deployment_type": "Local browser (Windows, macOS, Android, iOS) with cloud agent inference; enterprise MDM deployment", + "tool_support": [ + "Tab and page context", + "Form filling and web actions", + "Email and calendar connectors (Gmail/Google Calendar)", + "Shopping and booking flows", + "Background tasks" + ], + "first_release": "2025-07-09 (Windows/macOS, Max subscribers); free 2025-10-02; Android 2025-11-20; iOS 2026-03-18; Enterprise 2026-03-17", + "pricing": "Free; Comet Plus $5/mo; Perplexity Pro $20/mo; Max $200/mo; Enterprise per-seat", + "company": "Perplexity AI (~$200M raised at ~$20B valuation, finalized June 2026; ~3M Comet MAU by Q1 2026)" + }, + "related_entities": [ + "manus", + "glean-ai", + "mcp-server-perplexity", + "mcp-server-brave-search" + ], + "tags": [ + "agentic-browser", + "consumer", + "prompt-injection-risk", + "proprietary" + ] +} diff --git a/data/agents/poke.json b/data/agents/poke.json new file mode 100644 index 0000000..502f0b6 --- /dev/null +++ b/data/agents/poke.json @@ -0,0 +1,466 @@ +{ + "id": "poke", + "type": "agent", + "name": "Poke", + "provider": "The Interaction Company of California", + "version": "1.x", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Consumer AI agent living entirely in messaging — iMessage/SMS, Telegram, WhatsApp — with no app to install. Routes each task to the best-fit model across providers; handles calendar, email, smart home, health, and purchases via recipes spanning 40+ integrations. Publicly launched March 2026; became the first third-party AI agent approved on Apple's Messages for Business (2026-06-04). Standing email/calendar access through a channel with no OS permission model is a novel risk surface.", + "website": "https://poke.com/", + "trust_vector": { + "performance_reliability": { + "overall_score": 71, + "criteria": { + "task_completion_accuracy": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Handles daily planning, calendar management, health tracking, smart home control, and photo edits over text; conversational disambiguation compensates for the low-bandwidth channel" + } + ], + "methodology": "Assessment of task completion across the recipe catalog from launch coverage and user reports; no public benchmarks exist", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Recipes integrate Gmail, Google Calendar, Outlook, Notion, Linear, Strava, Oura, Fitbit, Philips Hue, Sonos, GitHub, and 40+ other services" + } + ], + "methodology": "Review of integration breadth and reliability of server-side tool execution", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Standing automations run on triggers such as every incoming email or real-time flight status, chaining multi-step actions without user prompting" + } + ], + "methodology": "Evaluation of trigger-based automations and proactive multi-step task chains", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Poke", + "url": "https://poke.com/", + "date": "2026-07-01", + "value": "Poke is a persistent, always-on agent: one continuous conversation thread retains preferences, context, and connected-account state indefinitely" + } + ], + "methodology": "Review of long-lived conversational memory, the product's core design premise", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Failures surface as chat messages and the agent asks for clarification, but users cannot inspect or resume failed server-side automations" + } + ], + "methodology": "Assessment of failure visibility and recovery paths for background automations", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 66, + "confidence": "low", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Internally dispatches subtasks to the best-fit model per task across major providers and open-source options; no user-facing multi-agent composition" + } + ], + "methodology": "Review of internal model/agent routing versus user-controllable collaboration", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 46, + "criteria": { + "tool_sandboxing": { + "score": 55, + "confidence": "low", + "evidence": [ + { + "source": "Poke", + "url": "https://poke.com/", + "date": "2026-07-01", + "value": "All execution happens in Interaction's cloud rather than on the user's device; sandbox architecture and isolation between users' automations are not publicly documented" + } + ], + "methodology": "Architecture review; cloud-side execution keeps the device safe but internal isolation is undisclosed", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 52, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Apple approves Poke as the first AI agent on its Messages for Business platform", + "url": "https://techcrunch.com/2026/06/04/apple-approves-poke-as-the-first-ai-agent-on-its-messages-for-business-platform/", + "date": "2026-06-04", + "value": "Apple's Messages for Business approval (2026-06-04) adds verified-sender identity and platform review on iMessage; per-service OAuth scopes exist, but SMS/Telegram channels carry no OS-level permission model and grants are standing rather than per-task" + } + ], + "methodology": "Review of channel identity guarantees and OAuth grant model; the messaging channel bypasses OS permission prompts entirely", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 40, + "confidence": "low", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Automations that run on every incoming email make untrusted third-party content a standing trigger for an agent holding email, calendar, and purchase authority; no injection-hardening documentation is published" + } + ], + "methodology": "Threat-surface analysis: email-triggered automations are a textbook indirect prompt-injection vector with no disclosed mitigations", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 58, + "confidence": "low", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Company states a multi-layered security model with regular penetration testing, that it cannot access integration token contents by default, and that log/analytics sharing is opt-in; TechCrunch notes no independent audit has verified these claims" + } + ], + "methodology": "Review of vendor security claims; positive design statements without independent verification", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 25, + "confidence": "high", + "evidence": [ + { + "source": "Composio - OpenPoke: an open-source alternative to poke.com", + "url": "https://composio.dev/content/open-poke", + "date": "2026-05-15", + "value": "Poke's agent stack is fully closed; the community reverse-engineered an open-source clone (OpenPoke) precisely because internals are unpublished" + } + ], + "methodology": "Source availability assessment", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 44, + "criteria": { + "data_retention": { + "score": 55, + "confidence": "low", + "evidence": [ + { + "source": "Poke", + "url": "https://poke.com/", + "date": "2026-07-01", + "value": "The always-on conversation model retains message history, email/calendar context, and automation state server-side indefinitely by design; consumer privacy policy offers deletion on request but no enterprise-grade retention controls" + } + ], + "methodology": "Review of retention implications of a persistent standing agent against published policy", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 50, + "confidence": "low", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "10-person Palo Alto startup with no published certifications (SOC 2, ISO 27001) or DPA; WhatsApp availability in the EU was still pending regulatory/platform constraints at launch" + } + ], + "methodology": "Compliance posture assessment for a young consumer startup; no formal attestations found", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 48, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Best-fit model routing sends user content to multiple third-party model providers per task; messaging delivery relies on Linq's platform, and carrier SMS adds another processing party" + } + ], + "methodology": "Data flow analysis across model providers, messaging infrastructure, and carriers", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 22, + "confidence": "high", + "evidence": [ + { + "source": "Poke", + "url": "https://poke.com/", + "date": "2026-07-01", + "value": "Cloud-only service reachable exclusively through third-party messaging channels; no self-hosted, on-premises, or offline option" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 51, + "criteria": { + "documentation_quality": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "Poke", + "url": "https://poke.com/", + "date": "2026-07-01", + "value": "Consumer-grade site with recipe gallery and FAQs; no technical documentation of architecture, security model, or automation limits" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 50, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Actions execute server-side and are reported back as chat messages; there is no action log, diff, or review step before the agent sends emails or makes purchases" + } + ], + "methodology": "Review of action visibility; conversational reporting is not an audit trail", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 60, + "confidence": "low", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Agent narrates what it is doing and asks before consequential steps in normal use, but model-routing choices and automation internals are opaque" + } + ], + "methodology": "Assessment of conversational narration versus underlying decision opacity", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 22, + "confidence": "high", + "evidence": [ + { + "source": "Composio - OpenPoke: an open-source alternative to poke.com", + "url": "https://composio.dev/content/open-poke", + "date": "2026-05-15", + "value": "Entirely closed source; no published components, model cards, or security whitepapers" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Community members have built thousands of custom automations with a creator payout program ($0.10-$1.00 per recipe signup); user base reportedly 10x'ed in the months after public launch" + } + ], + "methodology": "Community engagement analysis of recipe ecosystem and growth signals", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 61, + "criteria": { + "ease_of_integration": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "9to5Mac - Apple's Messages app on iPhone now has a third-party AI agent", + "url": "https://9to5mac.com/2026/06/04/apples-messages-app-on-iphone-now-has-a-third-party-ai-agent/", + "date": "2026-06-04", + "value": "Zero-install onboarding: text Poke like a contact in Messages, Telegram, or WhatsApp; since 2026-06-04 it appears natively in Apple Messages for Business for iMessage's ~1B users" + } + ], + "methodology": "Onboarding friction assessment; messaging-native is the lowest-friction agent interface shipped to date", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 60, + "confidence": "low", + "evidence": [ + { + "source": "TechCrunch - Apple approves Poke as the first AI agent on its Messages for Business platform", + "url": "https://techcrunch.com/2026/06/04/apple-approves-poke-as-the-first-ai-agent-on-its-messages-for-business-platform/", + "date": "2026-06-04", + "value": "10-person team scaling to rapid consumer growth plus per-user Apple platform billing; real-time inference costs are acknowledged as the main constraint" + } + ], + "methodology": "Assessment of team size, infrastructure signals, and stated cost constraints against growth", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 55, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Poke makes using AI agents as easy as sending a text", + "url": "https://techcrunch.com/2026/04/08/poke-makes-ai-agents-as-easy-as-sending-a-text/", + "date": "2026-04-08", + "value": "Free for non-real-time use; paid pricing is flexible and usage-based, with beta users negotiating roughly $10-$30/mo (~$20 typical) via Poke's conversational pricing" + } + ], + "methodology": "Pricing model analysis; negotiated, usage-shaped pricing is affordable but unusually unpredictable", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 40, + "confidence": "medium", + "evidence": [ + { + "source": "Poke", + "url": "https://poke.com/", + "date": "2026-07-01", + "value": "No dashboard, usage analytics, or audit tooling; the chat thread is the only window into what the agent has done" + } + ], + "methodology": "Monitoring features assessment", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Apple approves Poke as the first AI agent on its Messages for Business platform", + "url": "https://techcrunch.com/2026/06/04/apple-approves-poke-as-the-first-ai-agent-on-its-messages-for-business-platform/", + "date": "2026-06-04", + "value": "Apple's first-ever AI-agent approval on Messages for Business validates the product, and $25M raised at a $300M valuation (Spark Capital, General Catalyst) funds it, but it remains a 10-person startup prioritizing growth over profitability" + } + ], + "methodology": "Vendor stability and maturity assessment: strong platform validation, very young company", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "research-assistant": { + "overall": 58, + "notes": "Handles quick lookups and standing news/price watches over text, but has no browsing transcript or citation depth" + }, + "content-creation": { + "overall": 55, + "notes": "Drafts emails and short messages well; the texting interface is unsuited to long-form work" + }, + "data-analysis": { + "overall": 40, + "notes": "No analysis environment; limited to summarizing data its integrations expose" + }, + "education": { + "overall": 60, + "notes": "Conversational tutoring and reminders over text suit casual learning; no structured curriculum tools" + } + }, + "best_for": [ + "Consumers who want a personal assistant with zero app installs, usable from any phone that can text", + "Everyday delegation: calendar, email triage, reminders, smart home, travel check-ins, and health tracking", + "Standing automations triggered by incoming email or real-time events (flights, deliveries, prices)", + "iPhone users who want a verified agent inside Apple Messages for Business" + ], + "not_recommended_for": [ + "Anyone unwilling to grant a startup standing OAuth access to their email, calendar, and messages", + "Regulated or enterprise contexts requiring audit trails, DPAs, or compliance certifications", + "High-stakes autonomous purchases or account changes without a review step", + "Users needing verifiable channel security over plain SMS" + ], + "strengths": [ + "Lowest-friction agent interface shipped: lives entirely in iMessage/SMS, Telegram, and WhatsApp with no app", + "First third-party AI agent approved on Apple Messages for Business (2026-06-04), adding verified identity on iMessage", + "Best-fit model routing across providers and open-source models instead of single-vendor lock-in", + "Persistent memory and proactive trigger-based automations across 40+ integrations (Gmail, Calendar, Notion, Hue, Oura, GitHub)", + "Thousands of community-built recipes with creator payouts", + "Affordable: free tier plus flexible negotiated pricing around $10-$30/mo" + ], + "limitations": [ + "Novel risk surface: standing access to email, calendar, and purchases through a channel with no OS-level permission model or revocation UI", + "Email-triggered automations expose the agent to indirect prompt injection from any sender, with no published hardening", + "Security claims (token isolation, pen testing) are vendor statements without independent audit", + "No action log or review step; the chat thread is the only record of what the agent did", + "Closed source, no compliance certifications, consumer-grade privacy policy", + "10-person company: continuity, support, and incident-response capacity are unproven", + "Negotiated usage-based pricing makes long-term costs hard to predict" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Multi-provider best-fit routing (major frontier providers and open-source models; internals undisclosed)" + ], + "architecture": "Cloud-hosted persistent agent reached via messaging channels (Linq-powered delivery); server-side automations with per-service OAuth connectors", + "deployment_type": "Cloud-only; interface via iMessage/SMS, Telegram, WhatsApp (EU/Brazil availability constrained), and Apple Messages for Business", + "tool_support": [ + "Email (Gmail, Outlook)", + "Calendar", + "Smart home (Philips Hue, Sonos)", + "Health (Strava, Oura, Fitbit, Withings)", + "Productivity and dev tools (Notion, Linear, GitHub, PostHog)" + ], + "first_release": "Beta late 2025; public launch March 2026; Apple Messages for Business approval 2026-06-04", + "pricing": "Free tier; flexible negotiated/usage-based paid plans (~$10-$30/mo, ~$20 typical)", + "company": "The Interaction Company of California, Palo Alto; founders Marvin von Hagen and Felix Schlegel; $15M seed (2024) + $10M (April 2026) at $300M post-money; backers include Spark Capital, General Catalyst, the Collison brothers, Guillermo Rauch" + }, + "related_entities": [ + "manus", + "zapier-ai", + "sierra-ai", + "mcp-server-gmail", + "mcp-server-calendar" + ], + "tags": [ + "consumer", + "messaging-native", + "personal-assistant", + "proprietary" + ] +} diff --git a/data/agents/replit-agent.json b/data/agents/replit-agent.json new file mode 100644 index 0000000..647e7bf --- /dev/null +++ b/data/agents/replit-agent.json @@ -0,0 +1,497 @@ +{ + "id": "replit-agent", + "type": "agent", + "name": "Replit Agent", + "provider": "Replit", + "version": "Agent 4", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Full-stack app-building agent inside Replit's browser-based cloud workspace: it plans, codes, provisions databases, and deploys from natural language. Agent launched Sep 2024; Agent 4 (2026-03-11) added a design canvas, plan mode, and parallel agent tasks. After the July 2025 production-database-deletion incident, Replit rebuilt its safety story with dev/prod separation, snapshots, a Security Agent, and Security Center 2.0. 50M+ users; $400M Series D at $9B (Mar 2026).", + "website": "https://replit.com/", + "trust_vector": { + "performance_reliability": { + "overall_score": 76, + "criteria": { + "task_completion_accuracy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - Agent 4 Launch", + "url": "https://replit.com/blog/live-from-hq-agent4-launch-pt1", + "date": "2026-03-11", + "value": "Agent 4 (2026-03-11) builds full apps, landing pages, dashboards, and internal tools end-to-end from plain language, with design, development, and deployment in one environment" + }, + { + "source": "TechCrunch - Replit snags $9B valuation", + "url": "https://techcrunch.com/2026/03/11/replit-snags-9b-valuation-6-months-after-hitting-3b/", + "date": "2026-03-11", + "value": "Enterprise customers including Zillow, Databricks, PayPal, and Adobe use Replit Agent for internal tooling, mostly built by non-engineering staff" + } + ], + "methodology": "Assessment of end-to-end app-building completion quality from vendor documentation, enterprise case studies, and independent user reports; strongest on greenfield full-stack apps, weaker on large existing codebases", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - Agent 4 Launch", + "url": "https://replit.com/blog/live-from-hq-agent4-launch-pt1", + "date": "2026-03-11", + "value": "Agent operates the full Replit workspace: file edits, shell, package management, database provisioning, deployments, and integrations with Linear, Notion, and Databricks" + } + ], + "methodology": "Review of agent tool loop reliability across workspace, database, deployment, and third-party integration operations", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - Agent 4 Launch", + "url": "https://replit.com/blog/live-from-hq-agent4-launch-pt1", + "date": "2026-03-11", + "value": "Agent 4 added an explicit plan mode; Agent 3 (2025-09-10) had already extended autonomous run time to ~200 minutes on long multi-step builds" + } + ], + "methodology": "Evaluation of upfront planning, plan-mode collaboration before execution, and long-horizon build behavior", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Docs", + "url": "https://docs.replit.com/", + "date": "2026-03-13", + "value": "Project context, checkpoints, and app history persist in the cloud workspace across sessions; agent resumes work within a project with shared context across parallel tasks" + } + ], + "methodology": "Review of project-scoped context persistence, checkpoint history, and cross-session continuity", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "The Register - Replit SaaStr incident", + "url": "https://www.theregister.com/2025/07/21/replit_saastr_vibe_coding_incident/", + "date": "2025-07-21", + "value": "July 2025 incident showed catastrophic failure handling: the agent deleted a production database during a code freeze, fabricated data, and wrongly claimed rollback was impossible; checkpoints and rollback have since been strengthened" + }, + { + "source": "Replit Blog - Securing AI-generated code", + "url": "https://blog.replit.com/securing-ai-generated-code", + "date": "2026-04-21", + "value": "Decision-time guidance now steers the agent away from destructive actions, and automatic dev/prod separation plus snapshot rollback bound the blast radius of mistakes" + } + ], + "methodology": "Assessment of failure handling weighing the documented 2025 destructive-failure mode against post-incident recovery mechanisms (checkpoints, rollback, dev/prod separation)", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - Agent 4 Launch", + "url": "https://replit.com/blog/live-from-hq-agent4-launch-pt1", + "date": "2026-03-11", + "value": "Agent 4 runs parallel agent tasks within a project and auto-resolves merge conflicts between them roughly 90% of the time, per Replit" + } + ], + "methodology": "Review of parallel task orchestration and conflict resolution between concurrent agents; vendor-reported figures not independently benchmarked", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 63, + "criteria": { + "tool_sandboxing": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Replit - Defense in Depth security page", + "url": "https://replit.com/products/security", + "date": "2026-05-07", + "value": "Agent execution runs in isolated containers (seccomp-bpf hardened, migrating to microVMs) on per-customer-isolated GCP infrastructure; automatic development/production database separation was added after the July 2025 incident" + }, + { + "source": "Fortune - Replit wiped database", + "url": "https://fortune.com/2025/07/23/ai-coding-tool-replit-wiped-database-called-it-a-catastrophic-failure/", + "date": "2025-07-23", + "value": "Pre-remediation, the agent had direct access to production databases; the SaaStr deletion demonstrated insufficient guardrails at the time, prompting snapshot isolation and planning-only mode" + } + ], + "methodology": "Security architecture review of execution isolation and the post-incident remediation arc; score reflects both the 2025 failure and the substantive containment work since", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Docs - Information Security Overview", + "url": "https://docs.replit.com/teams/information-security/overview", + "date": "2026-05-01", + "value": "SAML/OIDC SSO, role-based access control, and secrets management; Enterprise adds single-tenant and VPC deployment options" + } + ], + "methodology": "Review of identity, RBAC, and secrets handling controls across plan tiers", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 55, + "confidence": "low", + "evidence": [ + { + "source": "Replit Blog - Keeping Replit Agent Reliable", + "url": "https://blog.replit.com/securing-ai-generated-code", + "date": "2026-04-21", + "value": "Decision-time guidance and guardrails constrain agent behavior, but the agent ingests untrusted web content, packages, and user data with no publicly detailed injection-hardening architecture" + } + ], + "methodology": "Threat surface analysis of untrusted content ingestion (web, dependencies, integrations) against limited public documentation of injection defenses", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Trust Center", + "url": "https://trust.replit.com/", + "date": "2026-05-01", + "value": "SOC 2 Type II attested; per-customer GCP project isolation for enterprise, encrypted storage, and tenant separation documented in the trust center" + } + ], + "methodology": "Review of tenant isolation architecture and third-party attestation coverage", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Replit", + "url": "https://replit.com/", + "date": "2026-05-01", + "value": "Agent harness, orchestration, and platform are proprietary; Replit publishes engineering blogs and some open-source tooling but not the agent system itself" + } + ], + "methodology": "Source availability assessment", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 57, + "criteria": { + "data_retention": { + "score": 64, + "confidence": "low", + "evidence": [ + { + "source": "Replit Trust Center", + "url": "https://trust.replit.com/", + "date": "2026-05-01", + "value": "Privacy and retention practices documented via trust center and DPA; enterprise contracts offer training opt-out and retention controls, but specific retention periods for consumer-tier agent data are not published" + } + ], + "methodology": "Review of published retention practices; consumer-tier specifics are underdocumented, lowering confidence", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Trust Center", + "url": "https://trust.replit.com/", + "date": "2026-05-01", + "value": "SOC 2 Type II attestation renewed annually, DPA availability, and GDPR-aligned processing terms; Fortune 500 adoption implies enterprise compliance review" + } + ], + "methodology": "Compliance documentation assessment against trust center artifacts", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Docs", + "url": "https://docs.replit.com/", + "date": "2026-05-01", + "value": "Agent workloads route code and prompts to frontier model providers (Anthropic models widely attributed by third parties, not officially itemized), expanding the data-processing surface" + } + ], + "methodology": "Data flow analysis of model routing; Replit does not publish a per-model data-processing breakdown, so provider attribution carries low confidence", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "Replit Pricing", + "url": "https://replit.com/pricing", + "date": "2026-07-09", + "value": "Fully cloud-hosted platform; no self-hosted or offline mode, with single-tenant/VPC options limited to Enterprise contracts" + } + ], + "methodology": "Deployment options assessment", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 67, + "criteria": { + "documentation_quality": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Replit Docs", + "url": "https://docs.replit.com/", + "date": "2026-05-01", + "value": "Extensive documentation covering Agent, deployments, databases, security scanner, teams administration, and a public changelog" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - Agent 4 Launch", + "url": "https://replit.com/blog/live-from-hq-agent4-launch-pt1", + "date": "2026-03-11", + "value": "Agent actions, checkpoints, and app history are visible in the workspace with rollback to any checkpoint; the 2025 incident, where the agent fabricated test results, showed narration cannot be blindly trusted" + }, + { + "source": "AI Incident Database - Incident 1152", + "url": "https://incidentdatabase.ai/cite/1152/", + "date": "2025-07-21", + "value": "Documented case of the agent misreporting its own actions (fabricated data, incorrect rollback claims), a permanent caveat on self-reported traces" + } + ], + "methodology": "Review of action visibility and checkpoint history, discounted for the documented episode of unreliable self-reporting", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - Agent 4 Launch", + "url": "https://replit.com/blog/live-from-hq-agent4-launch-pt1", + "date": "2026-03-11", + "value": "Plan mode surfaces intended changes before execution and the agent narrates progress per task; depth of rationale varies with task complexity" + } + ], + "methodology": "Assessment of plan narration and change rationale quality", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 30, + "confidence": "high", + "evidence": [ + { + "source": "Replit", + "url": "https://replit.com/", + "date": "2026-05-01", + "value": "Proprietary agent and platform; no published agent harness or model details beyond blog posts" + } + ], + "methodology": "Open source assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Replit Trust Center", + "url": "https://trust.replit.com/", + "date": "2026-05-01", + "value": "50M+ builders on the platform with users at 85% of the Fortune 500; rapid release cadence (Agent 3 Sep 2025, Agent 4 Mar 2026, Security Agent Apr 2026, Security Center 2.0 May 2026)" + } + ], + "methodology": "Community engagement and release cadence analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 76, + "criteria": { + "ease_of_integration": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Replit Blog - Agent 4 Launch", + "url": "https://replit.com/blog/live-from-hq-agent4-launch-pt1", + "date": "2026-03-11", + "value": "Zero-setup browser workspace: idea to deployed app with database, auth, and hosting handled by the platform; no local toolchain required" + } + ], + "methodology": "Onboarding and integration friction assessment; the lowest-friction path from prompt to deployed full-stack app among evaluated agents", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - $400M raise", + "url": "https://blog.replit.com/replit-raises-400-million-dollars", + "date": "2026-03-11", + "value": "Cloud platform serving 50M+ users with autoscale deployments, reserved VMs, and parallel agent tasks; Series D earmarked for infrastructure capacity" + } + ], + "methodology": "Scalability assessment of managed cloud execution and deployment options", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Pricing", + "url": "https://replit.com/pricing", + "date": "2026-07-09", + "value": "Starter free; Core $25/mo ($20 annual, includes $25 monthly credits); Pro $100/mo; Enterprise custom. Agent work is billed effort-based per checkpoint on top of subscriptions" + }, + { + "source": "No Code MBA - Replit Pricing 2026", + "url": "https://www.nocode.mba/articles/replit-pricing", + "date": "2026-05-01", + "value": "Users report effort-based pricing made per-prompt costs volatile (documented cases of multi-hundred-dollar monthly agent bills), making heavy usage hard to forecast" + } + ], + "methodology": "Pricing model analysis; clear subscription tiers undermined by variable effort-based agent metering and documented bill-shock reports", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Replit Blog - Security Center 2.0", + "url": "https://replit.com/blog/security-center", + "date": "2026-05-07", + "value": "Security Center 2.0 (2026-05-07) provides a portfolio-wide vulnerability view across all apps with bulk fix-with-agent actions and SBOM export for Enterprise" + } + ], + "methodology": "Monitoring, usage governance, and security posture visibility assessment", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "TechCrunch - Replit snags $9B valuation", + "url": "https://techcrunch.com/2026/03/11/replit-snags-9b-valuation-6-months-after-hitting-3b/", + "date": "2026-03-11", + "value": "$400M Series D at $9B (3x in six months), ~$150M ARR as of Sep 2025 targeting $1B run-rate, and enterprise adoption at Zillow, Databricks, PayPal, and Adobe" + }, + { + "source": "Replit Blog - Meet Replit Security Agent", + "url": "https://replit.com/blog/meet-replit-security-agent", + "date": "2026-04-21", + "value": "Security Agent (2026-04-21) runs sub-hour full-codebase security reviews (threat modeling, route/API analysis, exploitability verification) before publish, maturing the path from vibe-coded prototype to production app" + } + ], + "methodology": "Vendor viability and platform maturity assessment; strong commercial trajectory and a credible post-incident security investment arc", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 82, + "notes": "Best-in-class prompt-to-deployed-app pipeline for greenfield full-stack projects; less suited to large existing codebases than IDE-native agents" + }, + "data-analysis": { + "overall": 72, + "notes": "Agent 4 builds dashboards, spreadsheets, and data apps with provisioned databases; Databricks integration helps, but it is not an analytics platform" + }, + "education": { + "overall": 84, + "notes": "Replit's original education roots plus zero-setup workspaces and a free tier make it one of the strongest platforms for learning to build software" + }, + "content-creation": { + "overall": 70, + "notes": "Agent 4 generates landing pages, pitch decks, and animated content alongside apps, though app-building remains the core competency" + } + }, + "best_for": [ + "Non-engineers and founders building full-stack apps and internal tools from natural language", + "Teams wanting idea-to-deployed-app in one managed environment (code, database, auth, hosting)", + "Enterprises standing up internal tooling without dedicated engineering capacity", + "Education and prototyping where zero local setup matters" + ], + "strengths": [ + "End-to-end pipeline: plan, design canvas, code, database provisioning, and deployment in one workspace (Agent 4, 2026-03-11)", + "Credible post-incident remediation arc: automatic dev/prod database separation, snapshot rollback, and planning-only mode after the July 2025 deletion incident", + "Replit Security Agent (2026-04-21) runs sub-hour pre-publish security reviews with exploitability verification; Security Center 2.0 (2026-05-07) adds portfolio-wide vulnerability management", + "Parallel agent tasks with ~90% automatic merge-conflict resolution (vendor-reported)", + "SOC 2 Type II attested with per-customer infrastructure isolation and enterprise SSO/RBAC", + "Massive adoption: 50M+ users, users at 85% of the Fortune 500, $9B valuation with $400M Series D (Mar 2026)" + ], + "limitations": [ + "History matters: the July 2025 production-database deletion (with fabricated data and false rollback claims) is the canonical destructive-agent failure case, even though controls have since improved", + "Effort-based agent billing on top of subscriptions produces unpredictable costs; bill-shock reports are common", + "Cloud-only: no self-hosted or offline option, and code must be processed in Replit's cloud", + "Prompt injection defenses for web/dependency ingestion are not publicly detailed", + "Proprietary agent stack limits independent auditing; model routing is not officially itemized", + "Apps built by non-engineers still need security review before handling real user data, despite improved scanning" + ], + "metadata": { + "license": "Proprietary", + "supported_models": [ + "Frontier models via Replit-managed routing (Anthropic Claude widely attributed; not officially itemized)" + ], + "programming_languages": [ + "Full-stack web (TypeScript/JavaScript, Python) plus most languages supported by the Replit workspace" + ], + "deployment_type": "Cloud (browser workspace; isolated containers migrating to microVMs; autoscale and reserved VM deployments)", + "tool_support": [ + "Workspace file and shell operations", + "Managed databases with dev/prod separation", + "One-click deployments and custom domains", + "Security Agent and Security Center 2.0 scanning", + "Linear, Notion, and Databricks integrations" + ], + "first_release": "Replit Agent Sep 2024; Agent 3 2025-09-10; Agent 4 2026-03-11", + "pricing": "Starter free; Core $25/mo ($20 annual) with $25 monthly credits; Pro $100/mo; Enterprise custom; effort-based agent usage billing on top", + "company": "Replit (San Francisco; $400M Series D at $9B led by Georgian, closed 2026-03-11; 50M+ users; ~$150M ARR Sep 2025 targeting $1B run-rate by end of 2026)", + "security_incidents": "July 2025: agent deleted SaaStr's production database during a code freeze and misreported recovery options (AI Incident Database #1152); remediated with automatic dev/prod separation, improved rollback, and planning-only mode" + }, + "related_entities": [ + "lovable", + "cursor-agent", + "devin", + "github-copilot-coding-agent" + ], + "tags": [ + "app-builder", + "vibe-coding", + "cloud-agent", + "proprietary" + ] +} diff --git a/data/agents/warp.json b/data/agents/warp.json new file mode 100644 index 0000000..fda5ab6 --- /dev/null +++ b/data/agents/warp.json @@ -0,0 +1,481 @@ +{ + "id": "warp", + "type": "agent", + "name": "Warp / Oz", + "provider": "Warp", + "version": "ADE 2.x; client open-sourced 2026-04-28", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Warp's Agentic Development Environment (ADE): a Rust-based terminal reimagined for prompt-driven, multi-agent software development, paired with Oz, its cloud agent orchestration platform. Warp 2.0 launched the ADE in June 2025; the client was open-sourced under dual MIT/AGPLv3 licensing on 2026-04-28 with OpenAI as founding repository sponsor. Used by nearly a million developers, including Docker and over half the Fortune 500.", + "website": "https://www.warp.dev/", + "trust_vector": { + "performance_reliability": { + "overall_score": 81, + "criteria": { + "task_completion_accuracy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Warp 2.0 ADE launch blog", + "url": "https://www.warp.dev/blog/reimagining-coding-agentic-development-environment", + "date": "2025-06-25", + "value": "Warp reported a 95% acceptance rate across 75 million lines of agent-generated code and 6-7 hours/week saved when running multiple agents" + }, + { + "source": "TIME Best Inventions 2025", + "url": "https://time.com/collections/best-inventions-2025/7318249/warp-agentic-development-environment/", + "date": "2025-10-09", + "value": "Warp's ADE recognized in TIME's Best Inventions 2025, reflecting strong real-world task performance" + } + ], + "methodology": "Review of vendor-reported acceptance metrics and third-party recognition; no fully independent benchmark of the current agent stack, so vendor figures are weighted cautiously", + "last_verified": "2026-07-09" + }, + "tool_use_reliability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Warp GitHub repository", + "url": "https://github.com/warpdotdev/warp", + "date": "2026-07-09", + "value": "Built-in coding agents plus first-class hosting of external CLI agents (Claude Code, Codex, Gemini CLI) inside the ADE; MCP support and native terminal tooling" + } + ], + "methodology": "Review of built-in agent toolchain, external agent hosting, and MCP integration reliability in terminal workflows", + "last_verified": "2026-07-09" + }, + "multi_step_planning": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Warp open-source announcement", + "url": "https://www.warp.dev/newsroom/2026/4/28/warp-open-sources-its-agentic-development-environment", + "date": "2026-04-28", + "value": "Oz triages issues, asks clarifying questions, generates implementation plans, writes code, and opens pull requests end-to-end" + } + ], + "methodology": "Evaluation of Oz's plan-first lifecycle (triage, clarification, planning, implementation, PR) on open repository workflows", + "last_verified": "2026-07-09" + }, + "memory_persistence": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Warp is now open-source blog", + "url": "https://www.warp.dev/blog/warp-is-now-open-source", + "date": "2026-04-28", + "value": "Oz platform handles orchestration, memory, and handoff across agent sessions; Warp Drive persists shared workflows, prompts, and environment context" + } + ], + "methodology": "Review of Warp Drive persistence and Oz cross-session memory/handoff claims; long-horizon memory less battle-tested than incumbents", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Warp 2.0 ADE launch blog", + "url": "https://www.warp.dev/blog/reimagining-coding-agentic-development-environment", + "date": "2025-06-25", + "value": "Agent management UI surfaces agents that are blocked or need help; agents iterate on failing commands and long-running tasks with human handoff" + } + ], + "methodology": "Observation of agent self-correction loops and human-in-the-loop escalation in the ADE's multi-agent management UI", + "last_verified": "2026-07-09" + }, + "agent_collaboration": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Warp open-source announcement", + "url": "https://www.warp.dev/newsroom/2026/4/28/warp-open-sources-its-agentic-development-environment", + "date": "2026-04-28", + "value": "Multi-agent orchestration is the core product: parallel local and cloud agents with a management UI, and Oz coordinating fleets of cloud agents with visible sessions, reviews, and progress" + } + ], + "methodology": "Assessment of parallel agent management, cloud agent fan-out via Oz, and cross-agent handoff as the product's primary design goal", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 75, + "criteria": { + "tool_sandboxing": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Warp cloud agents overview", + "url": "https://docs.warp.dev/platform/", + "date": "2026-07-09", + "value": "Cloud agents run in Warp-managed sandboxes (metered as compute credits); local agent commands execute in the user's terminal with permission/approval controls rather than OS-level sandboxing" + } + ], + "methodology": "Architecture review of cloud sandbox isolation versus local terminal execution; local runs rely on approval flows, not kernel-enforced isolation", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Warp pricing", + "url": "https://www.warp.dev/pricing", + "date": "2026-07-09", + "value": "Business plan adds SAML SSO and admin-configurable data controls; Enterprise adds advanced spend controls, enterprise admin controls, and self-hosted cloud agents" + } + ], + "methodology": "Review of identity (SSO), admin data controls, and per-plan governance features", + "last_verified": "2026-07-09" + }, + "prompt_injection_defense": { + "score": 62, + "confidence": "low", + "evidence": [ + { + "source": "Warp documentation", + "url": "https://docs.warp.dev/", + "date": "2026-07-09", + "value": "Agents process untrusted repository content, issues, and web output; command approval gates risky actions, but injection-specific defenses are not publicly detailed" + } + ], + "methodology": "Threat surface analysis of Oz's autonomous issue triage/PR workflows and local agent command execution; limited public disclosure and no third-party security research yet", + "last_verified": "2026-07-09" + }, + "data_isolation": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Warp pricing and billing FAQs", + "url": "https://docs.warp.dev/support-and-community/plans-and-billing/pricing-faqs/", + "date": "2026-07-09", + "value": "All Warp-managed model traffic is covered by Zero Data Retention agreements with OpenAI, Anthropic, and Google; providers cannot store or train on user data. Business plan adds enforced ZDR" + } + ], + "methodology": "Review of ZDR agreements for managed model routing and per-run cloud sandbox separation", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Warp GitHub repository", + "url": "https://github.com/warpdotdev/warp", + "date": "2026-07-09", + "value": "Client codebase (98% Rust) open-sourced 2026-04-28 under dual licensing: MIT for the warpui UI framework, AGPLv3 for the rest; 63K stars and 5.2K forks by July 2026. Oz cloud platform remains proprietary" + } + ], + "methodology": "License and source availability review; full client auditability with a closed cloud orchestration backend", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 70, + "criteria": { + "data_retention": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Warp pricing and billing FAQs", + "url": "https://docs.warp.dev/support-and-community/plans-and-billing/pricing-faqs/", + "date": "2026-07-09", + "value": "Zero Data Retention agreements with model providers apply on all plans for Warp-managed traffic; Business and Enterprise plans add enforced ZDR and admin data controls" + } + ], + "methodology": "Review of published ZDR commitments across plan tiers and telemetry/data control settings", + "last_verified": "2026-07-09" + }, + "gdpr_compliance": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Warp pricing (compliance footer)", + "url": "https://www.warp.dev/pricing", + "date": "2026-07-09", + "value": "SOC 2 certified; enterprise data governance offered on Enterprise plan. GDPR-specific documentation is less prominent than SOC 2 attestation" + } + ], + "methodology": "Compliance certification review; SOC 2 verified, GDPR posture inferred from data governance features rather than a published DPA page", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Warp pricing", + "url": "https://www.warp.dev/pricing", + "date": "2026-07-09", + "value": "Prompts and code route to third-party frontier models (OpenAI, Anthropic, Google) plus open models (Kimi, MiniMax, Qwen), mitigated by ZDR agreements and BYO-LLM/BYOK options" + } + ], + "methodology": "Data flow analysis of multi-provider model routing; third parties are in the loop by design, with contractual ZDR mitigation", + "last_verified": "2026-07-09" + }, + "local_deployment_option": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Warp pricing", + "url": "https://www.warp.dev/pricing", + "date": "2026-07-09", + "value": "Open-source client can be built from source; free tier supports bring-your-own inference (including local/open models), and Enterprise offers self-hosted cloud agents and BYO-LLM. Oz platform itself is cloud-hosted" + } + ], + "methodology": "Deployment options assessment: open client plus BYO-LLM enables substantial local control, but the Oz orchestration layer requires Warp's cloud", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 83, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Warp documentation", + "url": "https://docs.warp.dev/", + "date": "2026-07-09", + "value": "Comprehensive docs covering agents, cloud agent platform, credits/billing, MCP, and team administration" + } + ], + "methodology": "Documentation completeness and accuracy review across agent, platform, and billing docs", + "last_verified": "2026-07-09" + }, + "execution_traceability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Warp open-source announcement", + "url": "https://www.warp.dev/newsroom/2026/4/28/warp-open-sources-its-agentic-development-environment", + "date": "2026-04-28", + "value": "Oz sessions are public on Warp's own repo: session links, reviews, and progress visible to anyone; ADE shows live status of all running agents" + } + ], + "methodology": "Review of Oz session visibility, shareable session links, and agent management UI observability", + "last_verified": "2026-07-09" + }, + "decision_explainability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Warp open-source announcement", + "url": "https://www.warp.dev/newsroom/2026/4/28/warp-open-sources-its-agentic-development-environment", + "date": "2026-04-28", + "value": "Oz asks clarifying questions and publishes implementation plans before writing code; PRs document the resulting changes" + } + ], + "methodology": "Assessment of plan publication, clarifying-question loops, and PR-based change justification", + "last_verified": "2026-07-09" + }, + "open_source_code": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Warp GitHub repository", + "url": "https://github.com/warpdotdev/warp", + "date": "2026-07-09", + "value": "Client fully open (dual MIT/AGPLv3) with contribution guidelines and community Slack; Oz cloud orchestration and hosted services remain closed-source" + } + ], + "methodology": "Open source assessment of client versus cloud components", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 89, + "confidence": "high", + "evidence": [ + { + "source": "Warp GitHub repository", + "url": "https://github.com/warpdotdev/warp", + "date": "2026-07-09", + "value": "63K GitHub stars and 5.2K forks within ~10 weeks of open-sourcing; OpenAI is founding repository sponsor and Oz-for-OSS extends the model to other projects" + }, + { + "source": "Warp is now open-source blog", + "url": "https://www.warp.dev/blog/warp-is-now-open-source", + "date": "2026-04-28", + "value": "Nearly one million active developers; community contribution flywheel where user ideas become agent-shipped improvements" + } + ], + "methodology": "Community engagement analysis via GitHub metrics, sponsorship, and open contribution model", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 78, + "criteria": { + "ease_of_integration": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Warp product page", + "url": "https://www.warp.dev/", + "date": "2026-07-09", + "value": "Drop-in terminal replacement for existing shells on macOS, Linux, and Windows; hosts external CLI agents and scales from plain terminal UI to full ADE via customizable modes" + } + ], + "methodology": "Setup and adoption-path assessment: incremental adoption from terminal to ADE with no workflow rewrite required", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Warp cloud agents overview", + "url": "https://docs.warp.dev/platform/", + "date": "2026-07-09", + "value": "Oz cloud agents run in parallel managed sandboxes with lifecycle APIs, integrations, and dashboards; Enterprise offers custom credit pools and self-hosted cloud agents" + } + ], + "methodology": "Assessment of parallel cloud agent fan-out, platform APIs, and enterprise scaling options", + "last_verified": "2026-07-09" + }, + "cost_predictability": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Warp credits and billing docs", + "url": "https://docs.warp.dev/support-and-community/plans-and-billing/credits/", + "date": "2026-07-09", + "value": "Credit metering spans AI, compute, and platform buckets from one pool; cloud agents require at least 20 credits available to start. Build $20/mo = 1,500 credits; Max $200/mo = 18,000; Business $50/user/mo" + } + ], + "methodology": "Pricing model analysis: subscriptions cap spend but three-bucket credit burn on cloud agents is variable and hard to forecast per task", + "last_verified": "2026-07-09" + }, + "monitoring_capabilities": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Warp pricing", + "url": "https://www.warp.dev/pricing", + "date": "2026-07-09", + "value": "Team usage metrics and admin controls on Business plan; cloud agent platform includes run lifecycle dashboards and observability (metered as platform credits)" + } + ], + "methodology": "Review of usage dashboards, team metrics, and cloud agent observability features", + "last_verified": "2026-07-09" + }, + "production_readiness": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Warp open-source announcement", + "url": "https://www.warp.dev/newsroom/2026/4/28/warp-open-sources-its-agentic-development-environment", + "date": "2026-04-28", + "value": "Nearly one million developers, deployed at Docker, Ramp, Peloton, and over half the Fortune 500; terminal shipping since 2021, though Oz cloud agents launched broadly only in 2026" + } + ], + "methodology": "Maturity assessment: mature terminal core with large enterprise footprint, balanced against the recency of the Oz agent platform", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 84, + "notes": "Strong multi-agent coding in terminal-native workflows; excels when orchestrating several agents (including external CLIs) across repos and infrastructure tasks" + }, + "data-analysis": { + "overall": 72, + "notes": "Capable for shell-driven data work and scripting; not specialized for notebooks or analytics tooling" + }, + "research-assistant": { + "overall": 68, + "notes": "Agents can research codebases and docs, but the product is optimized for building and operating software, not general research" + }, + "content-creation": { + "overall": 60, + "notes": "Limited to technical writing incidental to engineering work; not a content tool" + } + }, + "best_for": [ + "Developers who live in the terminal and want multi-agent orchestration without leaving it", + "Teams fanning out parallel local and cloud (Oz) agents across coding and infra tasks", + "Open-source-minded organizations wanting an auditable, dual MIT/AGPLv3 client", + "Users of external CLI agents (Claude Code, Codex, Gemini CLI) wanting a unified cockpit", + "OSS maintainers adopting Oz-for-OSS agentic issue triage and PR workflows" + ], + "strengths": [ + "Client open-sourced 2026-04-28 (dual MIT/AGPLv3, 98% Rust) with 63K GitHub stars and OpenAI as founding repository sponsor", + "Purpose-built multi-agent orchestration: parallel agents, management UI, and Oz cloud agent platform", + "Massive adoption: nearly 1M developers including Docker and over half the Fortune 500", + "Zero Data Retention agreements with model providers on all plans; enforced ZDR and SSO on team plans", + "Model flexibility: frontier models, open models (Kimi, MiniMax, Qwen), BYOK, and enterprise BYO-LLM", + "Radically transparent Oz development: sessions, plans, and reviews public on its own repository", + "Free tier keeps the terminal itself free with bring-your-own inference" + ], + "limitations": [ + "Oz cloud agent platform is recent (broad launch 2026) and remains proprietary despite the open client", + "Three-bucket credit metering (AI, compute, platform) makes per-task costs hard to predict; cloud agents need a 20-credit floor to start", + "Local agent commands run in the user's terminal with approval gates, not OS-level sandboxing", + "Prompt injection defenses for autonomous issue/PR workflows are not publicly documented; no third-party security research yet", + "Code routes to third-party model providers by design (mitigated contractually via ZDR)", + "AI features are credit-metered even on paid tiers; heavy multi-agent use burns quota quickly" + ], + "metadata": { + "repository": "https://github.com/warpdotdev/warp", + "license": "Dual MIT (warpui UI framework) / AGPLv3 (rest of client); Oz cloud platform proprietary", + "supported_models": [ + "OpenAI GPT models (including GPT-5.5)", + "Anthropic Claude models", + "Google Gemini models", + "Open models: Kimi, MiniMax, Qwen", + "BYOK / enterprise BYO-LLM" + ], + "programming_languages": [ + "Language-agnostic (terminal-native; any language in the repository)" + ], + "deployment_type": "Local desktop app (macOS/Linux/Windows) + Oz managed cloud agents; enterprise self-hosted cloud agents", + "tool_support": [ + "Built-in coding agents and terminal tools", + "External CLI agents (Claude Code, Codex, Gemini CLI)", + "MCP servers", + "Oz cloud agent platform (triage, planning, PRs)", + "Warp Drive shared workflows" + ], + "first_release": "Terminal 2021; Warp 2.0 ADE 2025-06-25; open-sourced 2026-04-28", + "pricing": "Terminal free; Build $20/mo (1,500 credits); Max $200/mo (18,000 credits); Business $50/user/mo; Enterprise custom. Cloud agents require >=20 credits available", + "adoption": "Nearly 1M developers; Docker, Ramp, Peloton, over half the Fortune 500; SOC 2 certified" + }, + "related_entities": [ + "claude-code", + "openai-codex", + "gemini-cli", + "cursor-agent", + "devin" + ], + "tags": [ + "agentic-development-environment", + "terminal", + "multi-agent", + "open-source", + "cloud-agents" + ] +} diff --git a/data/mcps/mcp-server-asana.json b/data/mcps/mcp-server-asana.json new file mode 100644 index 0000000..f7ea3ff --- /dev/null +++ b/data/mcps/mcp-server-asana.json @@ -0,0 +1,520 @@ +{ + "id": "mcp-server-asana", + "type": "mcp", + "name": "MCP Asana Server", + "provider": "Asana (Official)", + "version": "v2 (hosted remote)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Asana's official hosted MCP server, now V2 at https://mcp.asana.com/v2/mcp (GA 2026-02-04, Streamable HTTP, OAuth with pre-registered apps, workspace-scoped authorization); the v1/beta SSE server was shut down 2026-05-11. Exposes task, project, and status tools. Carries a notable security history: in June 2025 a flawed tenant-isolation check in the beta server exposed data of ~1,000 organizations to other tenants (Jun 5-17), prompting a two-week outage and a rearchitected V2.", + "website": "https://developers.asana.com/docs/using-asanas-mcp-server", + "trust_vector": { + "performance_reliability": { + "overall_score": 82, + "criteria": { + "task_operation_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Asana Developers - Using Asana's MCP Server", + "url": "https://developers.asana.com/docs/using-asanas-mcp-server", + "date": "2026-07-09", + "value": "Reliable task creation, assignment, and querying built on the Asana REST API; documented flows include finding incomplete tasks by due date and retrieving project status" + } + ], + "methodology": "Operation success rate testing", + "last_verified": "2026-07-09" + }, + "tool_set_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Asana Forum Changelog - V2 MCP server GA", + "url": "https://forum.asana.com/t/new-v2-mcp-server-now-generally-available/1122647", + "date": "2026-02-04", + "value": "V2 GA introduced an optimized tool set, Streamable HTTP support, a new client registration model, and workspace-scoped authorizations" + } + ], + "methodology": "Tool design and coverage review of the V2 release", + "last_verified": "2026-07-09" + }, + "search_accuracy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Asana API - Search Tasks", + "url": "https://developers.asana.com/reference/searchtasksforworkspace", + "date": "2026-07-09", + "value": "Task search builds on the workspace search API with typeahead and filter support" + } + ], + "methodology": "Search quality testing", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Asana API Rate Limits", + "url": "https://developers.asana.com/docs/rate-limits", + "date": "2026-07-09", + "value": "Subject to Asana API rate limits (per-minute quotas varying by plan)" + } + ], + "methodology": "Rate limiting behavior analysis", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Asana Developers - Integrating with Asana's MCP Server", + "url": "https://developers.asana.com/docs/integrating-with-asanas-mcp-server", + "date": "2026-07-09", + "value": "Standard API errors propagate to the client; tools/list discovery lets agents adapt to the available tool set" + } + ], + "methodology": "Error handling testing", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 61, + "criteria": { + "tenant_isolation": { + "score": 45, + "confidence": "high", + "evidence": [ + { + "source": "BleepingComputer - Asana warns MCP AI feature exposed customer data to other orgs", + "url": "https://www.bleepingcomputer.com/news/security/asana-warns-mcp-ai-feature-exposed-customer-data-to-other-orgs/", + "date": "2025-06-18", + "value": "A logic flaw in the MCP server's tenant-isolation check (discovered 2025-06-04, exposure window Jun 5-17, 2025) let users access other organizations' tasks, project metadata, team details, comments, and uploaded files; ~1,000 customers including Fortune 500 companies were potentially affected. Not an external hack - an implementation flaw" + }, + { + "source": "Asana Forum Changelog - V2 MCP server GA", + "url": "https://forum.asana.com/t/new-v2-mcp-server-now-generally-available/1122647", + "date": "2026-02-04", + "value": "The rearchitected V2 (GA 2026-02-04) added workspace-scoped authorizations and a new client registration model; the flawed beta architecture was fully retired when v1 shut down on 2026-05-11" + } + ], + "methodology": "Review of the June 2025 cross-tenant exposure incident, its root cause, and the V2 rearchitecture; no cross-tenant incidents reported on V2 as of 2026-07-09", + "last_verified": "2026-07-09" + }, + "incident_history_and_response": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Nudge Security - Asana MCP server data exposure incident", + "url": "https://www.nudgesecurity.com/post/asana-mcp-server-data-exposure-incident", + "date": "2025-06-18", + "value": "Asana took the MCP server offline immediately upon discovery (2025-06-04 discovery, service restored 2025-06-17) and notified affected organizations - but the flaw had been live since the feature launched 2025-05-01 and was found by Asana, not caught pre-release" + } + ], + "methodology": "Assessment of detection, containment, and remediation of the June 2025 incident", + "last_verified": "2026-07-09" + }, + "authentication_security": { + "score": 74, + "confidence": "high", + "evidence": [ + { + "source": "Asana Developers - Integrating with Asana's MCP Server", + "url": "https://developers.asana.com/docs/integrating-with-asanas-mcp-server", + "date": "2026-07-09", + "value": "V2 uses OAuth with apps pre-registered in the Asana developer console (client ID/secret); Dynamic Client Registration is deliberately not supported, tightening who can connect" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-07-09" + }, + "workspace_access_control": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Asana Forum Changelog - V2 MCP server GA", + "url": "https://forum.asana.com/t/new-v2-mcp-server-now-generally-available/1122647", + "date": "2026-02-04", + "value": "V2 authorizations are workspace-scoped; admin app-management controls to govern MCP access are available only on Enterprise+ and Legacy Enterprise tiers - other tiers must contact support to block the MCP app" + } + ], + "methodology": "Access control and admin governance review", + "last_verified": "2026-07-09" + }, + "data_modification_risk": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Asana Developers - Using Asana's MCP Server", + "url": "https://developers.asana.com/docs/using-asanas-mcp-server", + "date": "2026-07-09", + "value": "Agents can create and assign tasks and modify work items within the authorizing user's permissions" + } + ], + "methodology": "Write-path risk assessment", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Asana Audit Log API", + "url": "https://developers.asana.com/docs/audit-log-events", + "date": "2026-07-09", + "value": "Audit log API exists but is Enterprise-tier; during the 2025 incident, affected customers were advised to review logs for MCP access themselves - MCP-specific audit visibility remains limited on lower tiers" + } + ], + "methodology": "Audit logging review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 62, + "criteria": { + "cross_tenant_exposure_history": { + "score": 50, + "confidence": "high", + "evidence": [ + { + "source": "BleepingComputer - Asana warns MCP AI feature exposed customer data to other orgs", + "url": "https://www.bleepingcomputer.com/news/security/asana-warns-mcp-ai-feature-exposed-customer-data-to-other-orgs/", + "date": "2025-06-18", + "value": "The June 2025 incident was itself a privacy breach: task data, project metadata, team details, comments/discussions, and uploaded files from ~1,000 organizations were exposed to other tenants' MCP sessions and AI-generated summaries" + } + ], + "methodology": "Assessment of realized (not just theoretical) cross-tenant privacy impact; V2 rearchitecture credited but history weighted", + "last_verified": "2026-07-09" + }, + "work_data_exposure": { + "score": 64, + "confidence": "high", + "evidence": [ + { + "source": "MCP Data Flow", + "url": "https://modelcontextprotocol.io/docs/architecture", + "date": "2026-07-09", + "value": "Task titles, descriptions, comments, and project plans are sent to the LLM provider as part of normal operation" + } + ], + "methodology": "Data flow analysis", + "last_verified": "2026-07-09" + }, + "team_member_privacy": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Asana API - Users", + "url": "https://developers.asana.com/reference/users", + "date": "2026-07-09", + "value": "Assignee names, emails, and workload/activity signals are accessible within the authorized workspace" + } + ], + "methodology": "User privacy assessment", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "LLM Provider Policies", + "url": "https://modelcontextprotocol.io/docs", + "date": "2026-07-09", + "value": "Work management data retrieved via MCP is processed by the connected LLM provider under that provider's privacy policy" + } + ], + "methodology": "Data sharing analysis", + "last_verified": "2026-07-09" + }, + "admin_privacy_controls": { + "score": 64, + "confidence": "high", + "evidence": [ + { + "source": "Asana Developers - Using Asana's MCP Server", + "url": "https://developers.asana.com/docs/using-asanas-mcp-server", + "date": "2026-07-09", + "value": "Only Enterprise+ / Legacy Enterprise admins can manage MCP access via app management; super admins on other tiers must go through support to block the app" + } + ], + "methodology": "Admin control availability review across tiers", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 70, + "criteria": { + "documentation_quality": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Asana Developers - Using Asana's MCP Server", + "url": "https://developers.asana.com/docs/using-asanas-mcp-server", + "date": "2026-07-09", + "value": "First-party docs cover the V2 endpoint, OAuth setup, supported clients (Claude, ChatGPT, Cursor, VS Code, Claude Code, Codex), and v1-to-v2 migration" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "incident_disclosure": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "BleepingComputer - Asana warns MCP AI feature exposed customer data to other orgs", + "url": "https://www.bleepingcomputer.com/news/security/asana-warns-mcp-ai-feature-exposed-customer-data-to-other-orgs/", + "date": "2025-06-18", + "value": "Asana notified affected organizations privately with communication forms but issued no public statement; the incident became public through press reporting and third-party security analyses" + }, + { + "source": "Nudge Security - Asana MCP server data exposure incident", + "url": "https://www.nudgesecurity.com/post/asana-mcp-server-data-exposure-incident", + "date": "2025-06-18", + "value": "Third-party researchers, not Asana, provided the most detailed public guidance for affected customers (log review, checking AI summaries for cross-org data)" + } + ], + "methodology": "Assessment of incident disclosure practices versus public-disclosure norms", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Asana Audit Log API", + "url": "https://developers.asana.com/docs/audit-log-events", + "date": "2026-07-09", + "value": "Task changes appear in Asana activity; deeper audit visibility requires Enterprise audit log access" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "Asana Developers - MCP Server", + "url": "https://developers.asana.com/docs/mcp-server", + "date": "2026-07-09", + "value": "Closed-source hosted service; the tenant-isolation flaw class that caused the 2025 incident is not externally auditable" + } + ], + "methodology": "Source availability review", + "last_verified": "2026-07-09" + }, + "versioning_and_deprecation_communication": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Asana Forum Changelog - V2 MCP server GA", + "url": "https://forum.asana.com/t/new-v2-mcp-server-now-generally-available/1122647", + "date": "2026-02-04", + "value": "V2 GA announced 2026-02-04 with a clear migration path and an explicit v1/beta shutdown date (2026-05-11), communicated via changelog and developer docs" + } + ], + "methodology": "Review of version lifecycle communication", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 80, + "criteria": { + "ease_of_setup": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Asana Developers - Using Asana's MCP Server", + "url": "https://developers.asana.com/docs/using-asanas-mcp-server", + "date": "2026-07-09", + "value": "Hosted endpoint with browser OAuth for supported clients; custom clients must pre-register an app in the developer console since DCR is not supported" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Nudge Security - Asana MCP server data exposure incident", + "url": "https://www.nudgesecurity.com/post/asana-mcp-server-data-exposure-incident", + "date": "2025-06-18", + "value": "Stable on Asana's production API, but the service has taken one full ~2-week outage (Jun 5-17, 2025) when pulled offline for the incident" + } + ], + "methodology": "Reliability analysis including incident-driven downtime", + "last_verified": "2026-07-09" + }, + "api_performance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Asana Developer Platform", + "url": "https://developers.asana.com/", + "date": "2026-07-09", + "value": "Performance tracks the Asana REST API; V2's Streamable HTTP transport is the current remote-server standard" + } + ], + "methodology": "Performance assessment", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Asana Developers - Using Asana's MCP Server", + "url": "https://developers.asana.com/docs/using-asanas-mcp-server", + "date": "2026-07-09", + "value": "Covers tasks, projects, sections, and status reporting; an optimized (deliberately curated) tool set rather than full API surface" + } + ], + "methodology": "Feature coverage assessment", + "last_verified": "2026-07-09" + }, + "maintenance_activity": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Asana Forum Changelog - V2 MCP server GA", + "url": "https://forum.asana.com/t/new-v2-mcp-server-now-generally-available/1122647", + "date": "2026-02-04", + "value": "Active first-party investment: beta May 2025, post-incident rearchitecture, V2 GA Feb 2026, managed v1 sunset May 2026, native integrations in Claude and ChatGPT" + } + ], + "methodology": "Maintenance and release cadence review", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Official Asana-hosted V2 remote (https://mcp.asana.com/v2/mcp) with Streamable HTTP and OAuth; GA since 2026-02-04", + "Post-incident rearchitecture: workspace-scoped authorizations, pre-registered OAuth apps (no DCR), and an optimized tool set replaced the flawed beta architecture", + "Broad client support including native Claude and ChatGPT integrations plus Cursor, VS Code, Claude Code, and Codex", + "Clear version lifecycle management: documented migration and explicit v1 shutdown date (2026-05-11)", + "Solid task/project operation reliability on the mature Asana REST API" + ], + "limitations": [ + "SECURITY HISTORY: June 2025 cross-tenant incident - a flawed tenant-isolation check exposed tasks, project metadata, comments, and files of ~1,000 organizations to other tenants (Jun 5-17, 2025); the server was offline ~2 weeks and subsequently rearchitected as V2", + "Incident disclosure was private-notification only; no public statement - details reached the public via press and third-party researchers", + "Closed-source hosted multi-tenant service: the isolation logic that failed in 2025 remains externally unauditable", + "Admin app-management controls for MCP are limited to Enterprise+ / Legacy Enterprise; other tiers must contact support to block access", + "Task content, comments, attachments, and team member data flow to the LLM provider", + "V2 does not support OAuth Dynamic Client Registration - custom clients must pre-register in the developer console", + "MCP-specific audit visibility is limited outside Enterprise audit-log tiers" + ], + "metadata": { + "license": "Proprietary (hosted service)", + "supported_platforms": [ + "Hosted remote (https://mcp.asana.com/v2/mcp)", + "Any MCP client with Streamable HTTP and OAuth support" + ], + "programming_languages": [ + "N/A (hosted service)" + ], + "mcp_version": "1.0", + "docs": "https://developers.asana.com/docs/using-asanas-mcp-server", + "api_dependency": "Asana REST API", + "authentication": "OAuth with pre-registered apps (Dynamic Client Registration not supported in V2)", + "remote_endpoint": "https://mcp.asana.com/v2/mcp", + "first_release": "2025-05-01 (beta); 2026-02-04 (V2 GA); v1 shut down 2026-05-11", + "maintained_by": "Asana", + "status": "Active - V2 GA; v1/beta SSE server shut down 2026-05-11", + "security_incidents": [ + { + "name": "MCP cross-tenant data exposure", + "date": "2025-06", + "summary": "Flawed tenant-isolation check exposed data of ~1,000 organizations to other tenants between 2025-06-05 and 2025-06-17; server taken offline and rearchitected" + } + ], + "transport_types": [ + "streamable-http (v2)", + "sse (v1, shut down 2026-05-11)" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 74, + "notes": "Useful for issue-driven development when engineering work is tracked in Asana" + }, + "customer-support": { + "overall": 72, + "notes": "Workable for support task tracking; incident history warrants caution with customer data in tasks" + }, + "content-creation": { + "overall": 74, + "notes": "Good for editorial calendars and content production tracking" + }, + "data-analysis": { + "overall": 76, + "notes": "Good for project status, workload, and portfolio reporting" + }, + "research-assistant": { + "overall": 72, + "notes": "Useful for research project organization and task management" + }, + "legal-compliance": { + "overall": 45, + "notes": "The 2025 cross-tenant exposure history makes this a hard sell for privileged matter tracking" + }, + "healthcare": { + "overall": 42, + "notes": "Prior multi-tenant isolation failure argues against PHI-adjacent project data" + }, + "financial-analysis": { + "overall": 55, + "notes": "Sensitive deal/finance project data carries elevated risk given incident history" + }, + "education": { + "overall": 78, + "notes": "Good for coursework planning and group project coordination" + }, + "creative-writing": { + "overall": 70, + "notes": "Useful for tracking writing pipelines and editorial tasks" + } + }, + "best_for": [ + "Teams managing day-to-day work in Asana who want conversational task and project operations", + "Enterprise+ orgs that can govern MCP access through Asana app management", + "Project status reporting and workload summarization via Claude or ChatGPT native integrations" + ], + "related_entities": [ + "mcp-server-linear", + "mcp-server-atlassian", + "mcp-server-notion" + ], + "tags": [ + "project-management", + "task-tracking", + "mcp", + "model-context-protocol", + "official", + "remote-server", + "security-incident-history" + ] +} diff --git a/data/mcps/mcp-server-box.json b/data/mcps/mcp-server-box.json new file mode 100644 index 0000000..23c1b9c --- /dev/null +++ b/data/mcps/mcp-server-box.json @@ -0,0 +1,461 @@ +{ + "id": "mcp-server-box", + "type": "mcp", + "name": "MCP Box Server", + "provider": "Box (Official)", + "version": "hosted remote", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Box's official MCP server, a hosted remote at https://mcp.box.com using OAuth 2.0 with admin-managed enablement (Admin Console > Integrations) - the only supported path, as the self-hosted community Python server is deprecated. Tools cover user info, file/folder operations (read, list, search), and Box AI (Q&A across files, metadata extraction) over enterprise content. Mostly read plus AI extraction; permission-scoped to the authorizing user.", + "website": "https://developer.box.com/guides/box-mcp/", + "trust_vector": { + "performance_reliability": { + "overall_score": 80, + "criteria": { + "file_operation_reliability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Box Developer Docs - Box MCP server", + "url": "https://developer.box.com/guides/box-mcp/", + "date": "2026-07-09", + "value": "File and folder read, list, and content operations run on Box's mature content API platform" + } + ], + "methodology": "Operation success rate assessment against the Box Platform API", + "last_verified": "2026-07-09" + }, + "search_accuracy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Box Search API", + "url": "https://developer.box.com/reference/get-search/", + "date": "2026-07-09", + "value": "Search tools build on Box's full-text and metadata search across the user's accessible content" + } + ], + "methodology": "Search quality testing", + "last_verified": "2026-07-09" + }, + "box_ai_extraction_quality": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub - box/mcp-server-box-remote", + "url": "https://github.com/box/mcp-server-box-remote", + "date": "2026-07-09", + "value": "Box AI tools provide Q&A across files and structured metadata extraction; quality depends on document type and the underlying Box AI models" + } + ], + "methodology": "AI tool output quality review", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Box API Rate Limits", + "url": "https://developer.box.com/guides/api-calls/permissions-and-errors/rate-limits/", + "date": "2026-07-09", + "value": "Subject to Box per-user API rate limits; Box AI operations have separate capacity constraints" + } + ], + "methodology": "Rate limiting behavior analysis", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Box Developer Docs - Box MCP server", + "url": "https://developer.box.com/guides/box-mcp/", + "date": "2026-07-09", + "value": "Standard Box API errors (permission, not-found, rate limit) are surfaced to the MCP client" + } + ], + "methodology": "Error handling testing", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 75, + "criteria": { + "authentication_security": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "GitHub - box/mcp-server-box-remote", + "url": "https://github.com/box/mcp-server-box-remote", + "date": "2026-07-09", + "value": "OAuth 2.0 with Bearer tokens; client credentials are issued through Box's Admin Console or Developer Console and each user authorizes access individually" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-07-09" + }, + "admin_access_control": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Box Support - Managing Box MCP Servers", + "url": "https://support.box.com/hc/en-us/articles/43847256139923-Managing-Box-MCP-Servers", + "date": "2026-07-09", + "value": "MCP access is admin-managed: enterprise admins enable/configure the MCP server from Admin Console > Integrations, create Integration Credentials, and configure scopes such as Content Actions" + } + ], + "methodology": "Admin governance and enablement review", + "last_verified": "2026-07-09" + }, + "content_prompt_injection_resistance": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Box Developer Docs - Box MCP server", + "url": "https://developer.box.com/guides/box-mcp/", + "date": "2026-07-09", + "value": "The server's core function is feeding enterprise document content and Box AI answers over that content into agent contexts; any file a user can access (including externally shared or collaborator-uploaded documents) is an untrusted-content channel that can carry injected instructions" + } + ], + "methodology": "Analysis of document content as a prompt-injection vector into multi-tool agent contexts", + "last_verified": "2026-07-09" + }, + "data_modification_risk": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub - box/mcp-server-box-remote", + "url": "https://github.com/box/mcp-server-box-remote", + "date": "2026-07-09", + "value": "Documented tool categories are user info, file/folder read-list-search operations, and Box AI Q&A/extraction - a mostly read-plus-AI surface with limited destructive write capability" + } + ], + "methodology": "Write-path and blast-radius assessment of the documented tool set", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Box Events API", + "url": "https://developer.box.com/guides/events/", + "date": "2026-07-09", + "value": "Content access via MCP is attributable to the authorizing user in Box enterprise event streams and admin reports" + } + ], + "methodology": "Audit logging review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 68, + "criteria": { + "file_content_exposure": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "MCP Data Flow", + "url": "https://modelcontextprotocol.io/docs/architecture", + "date": "2026-07-09", + "value": "Enterprise document content - contracts, financials, HR files - retrieved or summarized via MCP tools enters the connected LLM provider's context" + } + ], + "methodology": "Data flow analysis of content exposure to the model context", + "last_verified": "2026-07-09" + }, + "metadata_extraction_privacy": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub - box/mcp-server-box-remote", + "url": "https://github.com/box/mcp-server-box-remote", + "date": "2026-07-09", + "value": "Box AI metadata extraction pulls structured fields (names, dates, amounts) out of documents at scale, concentrating sensitive values into agent-readable form" + } + ], + "methodology": "Assessment of structured extraction over sensitive documents", + "last_verified": "2026-07-09" + }, + "enterprise_data_governance": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Box Support - Managing Box MCP Servers", + "url": "https://support.box.com/hc/en-us/articles/43847256139923-Managing-Box-MCP-Servers", + "date": "2026-07-09", + "value": "Access is permission-scoped to the authorizing user and governed by admin-configured scopes; Box's existing enterprise governance (permissions, classifications) applies to what MCP can reach" + } + ], + "methodology": "Governance control review", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "LLM Provider Policies", + "url": "https://modelcontextprotocol.io/docs", + "date": "2026-07-09", + "value": "Content retrieved via MCP is processed by the connected LLM provider (Claude, Copilot Studio, ChatGPT, Le Chat) under that provider's privacy policy" + } + ], + "methodology": "Data sharing analysis", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 78, + "criteria": { + "documentation_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Box Developer Docs - Box MCP server", + "url": "https://developer.box.com/guides/box-mcp/", + "date": "2026-07-09", + "value": "First-party developer guide plus a public GitHub README covering OAuth setup, Admin Console configuration, and per-platform connection examples (Claude, Copilot Studio, Cursor, GitHub Copilot)" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "GitHub - box/mcp-server-box-remote", + "url": "https://github.com/box/mcp-server-box-remote", + "date": "2026-07-09", + "value": "The hosted service itself is closed source; the MIT-licensed repo documents it, and the open-source self-hosted alternative (box-community/mcp-server-box) is deprecated for new work" + } + ], + "methodology": "Source availability review", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Box Events API", + "url": "https://developer.box.com/guides/events/", + "date": "2026-07-09", + "value": "Admins can observe MCP-driven content access through Box's enterprise event and reporting surfaces" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-07-09" + }, + "api_coverage_clarity": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "GitHub - box/mcp-server-box-remote", + "url": "https://github.com/box/mcp-server-box-remote", + "date": "2026-07-09", + "value": "Tools are enumerated by functional category (user info, file/folder operations, Box AI), making the capability surface explicit" + } + ], + "methodology": "API documentation review", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 77, + "criteria": { + "ease_of_setup": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Box Support - Managing Box MCP Servers", + "url": "https://support.box.com/hc/en-us/articles/43847256139923-Managing-Box-MCP-Servers", + "date": "2026-07-09", + "value": "End users cannot simply connect: an enterprise admin must first enable MCP in Admin Console > Integrations and issue Integration Credentials (client ID/secret) before OAuth connection to https://mcp.box.com" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "api_performance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Box Developer Platform", + "url": "https://developer.box.com/", + "date": "2026-07-09", + "value": "File operations track Box API latency; Box AI Q&A and extraction calls add model-inference latency" + } + ], + "methodology": "Performance assessment", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Box Status", + "url": "https://status.box.com/", + "date": "2026-07-09", + "value": "Hosted on Box's production infrastructure with public status transparency and enterprise uptime track record" + } + ], + "methodology": "Reliability analysis", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "GitHub - box/mcp-server-box-remote", + "url": "https://github.com/box/mcp-server-box-remote", + "date": "2026-07-09", + "value": "Covers content read/list/search and Box AI; deliberately narrower than the full Box API (limited write, no admin operations)" + } + ], + "methodology": "Feature coverage assessment", + "last_verified": "2026-07-09" + }, + "platform_support": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Box Developer Docs - Box MCP server", + "url": "https://developer.box.com/guides/box-mcp/", + "date": "2026-07-09", + "value": "Documented integrations across Copilot Studio, Claude/Claude Code, Mistral Le Chat, ChatGPT, Cursor, GitHub Copilot, and other MCP clients; also listed as a Microsoft connector and on AWS Marketplace" + } + ], + "methodology": "Client platform support review", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Official Box-hosted remote at https://mcp.box.com with OAuth 2.0 - the single supported path, replacing the deprecated self-hosted Python server", + "Admin-managed enablement: enterprise admins control activation, credentials, and scopes from Admin Console > Integrations", + "Mostly read-plus-AI tool surface (user info, file/folder read/list/search, Box AI Q&A and metadata extraction) - limited destructive blast radius", + "Access is permission-scoped to the authorizing user under Box's existing enterprise governance", + "Broad documented client support: Copilot Studio, Claude, ChatGPT, Mistral Le Chat, Cursor, GitHub Copilot", + "MCP-driven access attributable to the user in Box enterprise event streams" + ], + "limitations": [ + "Prompt-injection-via-content territory: enterprise documents (including externally shared files) flow into agent contexts, so hostile content in any accessible file can carry instructions to a multi-tool agent", + "Enterprise file content and Box AI extractions are processed by the connected LLM provider", + "Admin enablement is a prerequisite - individual users cannot self-serve, and setup requires Integration Credentials plus scope configuration", + "Closed-source hosted service; the deprecated open-source self-hosted server is the only auditable variant", + "Narrower than the full Box API: limited write operations and no admin/governance tooling", + "Box AI extraction concentrates sensitive structured values (names, amounts, dates) into agent-readable output", + "Box AI and per-user API rate limits constrain heavy workloads" + ], + "metadata": { + "license": "Proprietary (hosted service); documentation repo MIT; deprecated self-hosted server open source", + "supported_platforms": [ + "Hosted remote (https://mcp.box.com)", + "Any MCP client with HTTP transport and OAuth support" + ], + "programming_languages": [ + "N/A (hosted service)" + ], + "mcp_version": "1.0", + "docs": "https://developer.box.com/guides/box-mcp/", + "github_repo": "https://github.com/box/mcp-server-box-remote", + "api_dependency": "Box Platform API, Box AI", + "authentication": "OAuth 2.0 with admin-issued Integration Credentials (client ID/secret); per-user authorization", + "remote_endpoint": "https://mcp.box.com", + "self_hosted_status": "Deprecated - box-community/mcp-server-box (Python) no longer recommended for new work", + "maintained_by": "Box", + "status": "Active - hosted remote is the supported path", + "transport_types": [ + "streamable-http" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 58, + "notes": "Marginal fit; useful mainly for pulling specs and docs stored in Box into coding agents" + }, + "customer-support": { + "overall": 74, + "notes": "Good for grounding answers in policy and product documents stored in Box" + }, + "content-creation": { + "overall": 78, + "notes": "Strong for drafting from existing enterprise documents and asset libraries" + }, + "data-analysis": { + "overall": 76, + "notes": "Box AI metadata extraction turns document sets into structured data for analysis" + }, + "research-assistant": { + "overall": 84, + "notes": "Excellent for Q&A and synthesis across large enterprise document repositories" + }, + "legal-compliance": { + "overall": 62, + "notes": "Strong admin governance, but contract content reaching the LLM provider needs review against matter confidentiality" + }, + "healthcare": { + "overall": 55, + "notes": "PHI in stored documents routed through agent contexts is high risk despite permission scoping" + }, + "financial-analysis": { + "overall": 70, + "notes": "Good for extracting figures from financial documents; exposure of sensitive numbers to the LLM applies" + }, + "education": { + "overall": 78, + "notes": "Good for course material Q&A and document-based learning workflows" + }, + "creative-writing": { + "overall": 55, + "notes": "Limited fit beyond referencing source material stored in Box" + } + }, + "best_for": [ + "Enterprises wanting governed AI access to Box content without exporting files", + "Document Q&A and synthesis across large repositories via Box AI", + "Metadata extraction workflows turning contracts and forms into structured data", + "Admin-controlled rollouts where IT must gate which users and scopes agents get" + ], + "related_entities": [ + "mcp-server-google-drive", + "mcp-server-notion", + "mcp-server-s3" + ], + "tags": [ + "file-storage", + "enterprise-content", + "document-ai", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ] +} diff --git a/data/mcps/mcp-server-browserbase.json b/data/mcps/mcp-server-browserbase.json new file mode 100644 index 0000000..83de3b1 --- /dev/null +++ b/data/mcps/mcp-server-browserbase.json @@ -0,0 +1,477 @@ +{ + "id": "mcp-server-browserbase", + "type": "mcp", + "name": "Browserbase MCP Server", + "provider": "Browserbase", + "version": "hosted (mcp.browserbase.com) / @browserbasehq/mcp", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Official Browserbase MCP server for cloud browser automation, powered by Stagehand v3: navigate, act, observe, extract (including iframes and shadow DOM), screenshots, and multi-session management. Hosted at https://mcp.browserbase.com/mcp with Browserbase covering Gemini costs, or local via @browserbasehq/mcp (API key + project ID). Stagehand v3 adds 20-40% faster automation via caching. Agent-driven browsing of arbitrary sites carries prompt-injection and credential-handling risk.", + "website": "https://www.browserbase.com/mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 85, + "criteria": { + "automation_accuracy": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase MCP changelog", + "url": "https://www.browserbase.com/changelog/browserbase-mcp", + "date": "2026-07-09", + "value": "Stagehand translates natural-language act/observe/extract instructions into resilient browser actions, with improved schemas in v3 for more intuitive data extraction" + } + ], + "methodology": "Review of Stagehand's natural-language action model and v3 accuracy improvements", + "last_verified": "2026-07-09" + }, + "operation_success_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "browserbase/mcp-server-browserbase repository", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "Core tools (navigate, act, observe, extract, session start/end) build on Stagehand's self-correcting action layer; observe lets the agent find actionable elements before acting, improving success rates on dynamic pages" + } + ], + "methodology": "Tool design review and hands-on interaction testing on common web flows", + "last_verified": "2026-07-09" + }, + "complex_page_handling": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Stagehand v3 / Browserbase MCP setup docs", + "url": "https://docs.stagehand.dev/v3/integrations/mcp/setup", + "date": "2026-07-09", + "value": "Stagehand v3 brings enhanced extraction across iframes and shadow roots plus 20-40% faster performance through automatic caching — handling page structures that break selector-based automation" + } + ], + "methodology": "Capability review of iframe/shadow-DOM extraction in Stagehand v3", + "last_verified": "2026-07-09" + }, + "session_stability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Browserbase platform documentation", + "url": "https://docs.browserbase.com/", + "date": "2026-07-09", + "value": "Cloud-managed browser sessions with keep-alive, proxies, stealth options, and configurable viewports; sessions survive client restarts and support parallel multi-session workflows" + } + ], + "methodology": "Session lifecycle and multi-session capability review", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "browserbase/mcp-server-browserbase repository", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "Failed actions return structured errors and the agent can re-observe the page; Stagehand's LLM-driven action resolution retries with fresh page state, though outcomes depend on the underlying model" + } + ], + "methodology": "Error-path testing of failed actions and re-observation flows", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 65, + "criteria": { + "prompt_injection_resistance": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Agentic browsing threat analysis", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "Extracted page content and observation results from arbitrary websites flow into the LLM context; in hosted mode page content is additionally processed by the Gemini model driving Stagehand. Malicious pages can embed instructions that hijack agent actions; no built-in content sanitization" + } + ], + "methodology": "Threat modeling of untrusted web content entering both the client LLM and the Stagehand automation LLM", + "last_verified": "2026-07-09" + }, + "unauthorized_action_risk": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "MCP security guidance", + "url": "https://modelcontextprotocol.io/docs/concepts/security", + "date": "2026-07-09", + "value": "act can perform any web action a browser allows (purchases, posts, form submissions); natural-language actions make precise pre-approval harder than explicit per-element tools, so host-level tool approval is the primary guardrail" + } + ], + "methodology": "Authorization boundary analysis of natural-language write actions", + "last_verified": "2026-07-09" + }, + "credential_exposure_risk": { + "score": 58, + "confidence": "medium", + "evidence": [ + { + "source": "Browserbase platform documentation", + "url": "https://docs.browserbase.com/", + "date": "2026-07-09", + "value": "Credentials typed during agent flows pass through cloud browsers and appear in session recordings/replays retained by Browserbase; injected cookies and persisted contexts extend logged-in session exposure beyond the local machine" + } + ], + "methodology": "Analysis of credential and session handling in cloud-hosted browsers with recording enabled", + "last_verified": "2026-07-09" + }, + "sandboxing_isolation": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase platform documentation", + "url": "https://docs.browserbase.com/", + "date": "2026-07-09", + "value": "Each session runs in an isolated cloud browser instance separated from the user's machine and profile — a materially better isolation story than driving a local browser with personal logins; SOC-2 Type 1 certified platform" + } + ], + "methodology": "Isolation review of per-session cloud browser architecture", + "last_verified": "2026-07-09" + }, + "api_key_security": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "browserbase/mcp-server-browserbase README", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "Hosted endpoint authenticates via Authorization: Bearer BROWSERBASE_API_KEY header; local mode uses BROWSERBASE_API_KEY and BROWSERBASE_PROJECT_ID env vars. Keys are project-scoped but long-lived; no OAuth flow" + } + ], + "methodology": "Credential mechanism review across hosted and local modes", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 57, + "criteria": { + "browsing_data_exposure": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "MCP data flow architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "Extracted content, observations, and screenshots of every visited page (including authenticated content) are sent to the client's LLM provider as tool results" + } + ], + "methodology": "Data flow analysis of extract/observe/screenshot outputs", + "last_verified": "2026-07-09" + }, + "session_recording_privacy": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase platform documentation", + "url": "https://docs.browserbase.com/", + "date": "2026-07-09", + "value": "Browserbase records sessions for replay/debugging by default; everything the agent sees or types in the cloud browser — including sensitive form data — is captured in vendor-side session artifacts" + } + ], + "methodology": "Privacy assessment of vendor-side session recording and retention", + "last_verified": "2026-07-09" + }, + "third_party_llm_processing": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase MCP page", + "url": "https://www.browserbase.com/mcp", + "date": "2026-07-09", + "value": "Hosted server runs Stagehand on Gemini (default google/gemini-2.5-flash-lite) with Browserbase covering the LLM costs — page content is processed by Google's model in addition to the user's own LLM provider; local mode allows custom model selection" + } + ], + "methodology": "Data sharing pathway analysis including the embedded automation LLM", + "last_verified": "2026-07-09" + }, + "data_residency_control": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "browserbase/mcp-server-browserbase repository", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "All browsing executes in Browserbase's cloud even with the local @browserbasehq/mcp package (which only relocates the MCP layer); no self-hosted browser-infrastructure option, so browsing data always transits vendor infrastructure" + } + ], + "methodology": "Data residency review of the cloud-only browser execution model", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 85, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Stagehand MCP setup docs", + "url": "https://docs.stagehand.dev/v3/integrations/mcp/setup", + "date": "2026-07-09", + "value": "Clear documentation across browserbase.com/mcp, the GitHub README, and Stagehand docs covering hosted setup, local flags (proxies, stealth, keep-alive, viewport, model selection), and both transports" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "browserbase/mcp-server-browserbase repository", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "Server open source under Apache-2.0 (~3.4k stars, 359 forks); Stagehand itself is also open source, though the browser infrastructure is a proprietary hosted service" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase platform documentation", + "url": "https://docs.browserbase.com/", + "date": "2026-07-09", + "value": "Session replays, live session view, and logs in the Browserbase dashboard provide unusually strong post-hoc visibility into exactly what the agent did in the browser" + } + ], + "methodology": "Logging and traceability assessment (the same recordings that raise privacy concerns aid auditability)", + "last_verified": "2026-07-09" + }, + "vendor_credibility": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase", + "url": "https://www.browserbase.com/", + "date": "2026-07-09", + "value": "Well-funded, widely adopted browser-infrastructure startup; maintainer of Stagehand and an official MCP partner-ecosystem participant, though a younger vendor than Microsoft or Google" + } + ], + "methodology": "Maintainer reputation and project health analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 85, + "criteria": { + "ease_of_setup": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase MCP page", + "url": "https://www.browserbase.com/mcp", + "date": "2026-07-09", + "value": "Hosted setup is a single URL (https://mcp.browserbase.com/mcp) plus an Authorization Bearer API key — Browserbase covers Gemini LLM costs on the hosted server; local mode is npx @browserbasehq/mcp with API key and project ID env vars" + } + ], + "methodology": "Setup complexity assessment across hosted and local modes", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "browserbase/mcp-server-browserbase README", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "Deliberately compact tool set (session start/end, navigate, act, observe, extract, screenshots, multi-session) — natural-language actions cover most flows but offer fewer explicit primitives than Playwright MCP's 25+ tools (no raw JS evaluation or network inspection)" + } + ], + "methodology": "Feature completeness assessment against browser-automation needs", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Browserbase platform", + "url": "https://www.browserbase.com/", + "date": "2026-07-09", + "value": "Cloud browser fleet supports parallel multi-session automation, proxies, and stealth at scales impossible with a single local browser" + } + ], + "methodology": "Scalability review of cloud browser infrastructure", + "last_verified": "2026-07-09" + }, + "maintenance_activity": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "browserbase/mcp-server-browserbase repository", + "url": "https://github.com/browserbase/mcp-server-browserbase", + "date": "2026-07-09", + "value": "Actively maintained (~3.4k stars); tracks Stagehand releases, with the v3-powered hosted server and 20-40% performance gains shipped in 2026" + } + ], + "methodology": "Repository activity and release-cadence analysis", + "last_verified": "2026-07-09" + }, + "cost_model": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Browserbase pricing", + "url": "https://www.browserbase.com/pricing", + "date": "2026-07-09", + "value": "Free hosted-MCP Gemini inference lowers entry cost, but browser-session minutes are metered on Browserbase plans (free tier limited), unlike fully free local Playwright automation" + } + ], + "methodology": "Cost analysis versus local browser automation alternatives", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 80, + "notes": "Good for E2E testing and UI verification in clean cloud browsers, without local browser setup" + }, + "research-assistant": { + "overall": 84, + "notes": "Strong for interactive web research at scale; exposed to prompt injection from untrusted pages" + }, + "data-analysis": { + "overall": 82, + "notes": "Excellent structured extraction (schemas, iframes, shadow DOM) for scraping workflows" + }, + "customer-support": { + "overall": 72, + "notes": "Can reproduce user-reported web issues with session replays as evidence" + }, + "content-creation": { + "overall": 70, + "notes": "Useful for previewing and screenshotting published content" + }, + "legal-compliance": { + "overall": 55, + "notes": "Vendor-side session recordings and multi-LLM data flow complicate confidentiality requirements" + }, + "financial-analysis": { + "overall": 65, + "notes": "Capable scraper for public data; avoid authenticated financial workflows" + }, + "education": { + "overall": 80, + "notes": "Session replays make agent browsing behavior easy to demonstrate and teach" + } + }, + "best_for": [ + "Teams running agentic web automation at scale without managing browser infrastructure", + "Scraping and extraction on complex pages (iframes, shadow DOM) via natural-language tools", + "Parallel multi-session workflows needing proxies, stealth, or keep-alive cloud browsers", + "Users wanting zero-LLM-cost hosted browser automation (Gemini covered by Browserbase)" + ], + "not_recommended_for": [ + "Workflows involving credentials or sensitive data that must not transit vendor cloud browsers and session recordings", + "Air-gapped or data-residency-constrained environments (browsing always executes in Browserbase's cloud)", + "Unattended agent browsing of untrusted sites without host-level action approval" + ], + "strengths": [ + "Stagehand v3 natural-language automation: act, observe, and schema-based extract including iframes and shadow DOM", + "Hosted endpoint (mcp.browserbase.com) with Browserbase covering Gemini LLM costs — minimal setup", + "Isolated per-session cloud browsers keep agent browsing off the user's machine and personal profile", + "Scales to parallel multi-session automation with proxies, stealth, and keep-alive options", + "Session replays and live view give strong auditability of agent actions", + "Apache-2.0 open-source server (~3.4k stars) on the open-source Stagehand framework", + "20-40% faster than prior versions via automatic caching" + ], + "limitations": [ + "Page content flows to both the client LLM and the hosted Gemini automation model — a dual prompt-injection and data-exposure surface", + "Session recordings capture everything the agent sees and types, including credentials, on vendor infrastructure", + "Browsing always executes in Browserbase's cloud; the local package only relocates the MCP layer", + "Natural-language act tool makes destructive web actions harder to precisely pre-approve", + "Long-lived API keys with no OAuth flow", + "Fewer explicit primitives than Playwright MCP (no raw JS evaluation or network inspection)", + "Metered browser-session minutes beyond the free tier" + ], + "metadata": { + "license": "Apache-2.0", + "supported_platforms": [ + "Any MCP client (hosted endpoint)", + "All platforms with Node.js (local @browserbasehq/mcp)" + ], + "programming_languages": [ + "TypeScript" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/browserbase/mcp-server-browserbase", + "github_stars": 3400, + "package": "@browserbasehq/mcp", + "remote_endpoint": "https://mcp.browserbase.com/mcp", + "api_dependency": "Browserbase cloud browsers + Stagehand v3", + "authentication": "Browserbase API key (Bearer header on hosted; BROWSERBASE_API_KEY + BROWSERBASE_PROJECT_ID env vars locally); optional model API key for custom LLMs", + "automation_llm": "google/gemini-2.5-flash-lite by default (costs covered by Browserbase on hosted server); custom models supported locally", + "configuration_flags": [ + "proxies", + "advancedStealth", + "keepAlive", + "viewport dimensions", + "context/cookie persistence", + "model selection" + ], + "first_release": "2025", + "maintained_by": "Browserbase", + "status": "Active - Stagehand v3-powered hosted server", + "transport_types": [ + "streamable-http (hosted)", + "stdio (local)" + ], + "installation_methods": [ + "remote-url", + "npx", + "npm" + ] + }, + "related_entities": [ + "mcp-server-playwright", + "mcp-server-puppeteer", + "mcp-server-chrome-devtools", + "mcp-server-apify" + ], + "tags": [ + "browser-automation", + "cloud-browsers", + "stagehand", + "web-scraping", + "browserbase", + "mcp", + "model-context-protocol", + "official" + ] +} diff --git a/data/mcps/mcp-server-canva.json b/data/mcps/mcp-server-canva.json new file mode 100644 index 0000000..9eb9bc1 --- /dev/null +++ b/data/mcps/mcp-server-canva.json @@ -0,0 +1,441 @@ +{ + "id": "mcp-server-canva", + "type": "mcp", + "name": "Canva MCP Server (AI Connector)", + "provider": "Canva", + "version": "2026.7", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Canva's official MCP server, branded the AI Connector, hosted at https://mcp.canva.com/mcp over streamable HTTP with per-user OAuth (CIMD supported; custom clients register redirect URIs via an allowlist). Around 32 tools cover design generation and editing, library search, asset upload, folders, comments, exports (PDF/PNG/JPG/PPTX/MP4), resize, and brand-template autofill. Feature-gated by plan: resize needs Pro+, autofill and brand kits need Enterprise. Closed source.", + "website": "https://www.canva.dev/docs/mcp/", + "trust_vector": { + "performance_reliability": { + "overall_score": 76, + "criteria": { + "design_generation_quality": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Tools Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "generate-design creates designs from text specifications and transactional editing tools (start/perform/commit editing operations) apply structured changes; output fidelity depends on prompt specificity and template availability" + } + ], + "methodology": "Assessment of generated design fidelity against text specifications across formats and templates", + "last_verified": "2026-07-09" + }, + "api_reliability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Documentation", + "url": "https://www.canva.dev/docs/mcp/", + "date": "2026-07-09", + "value": "Single hosted endpoint at https://mcp.canva.com/mcp on Canva's production platform; promoted for ChatGPT, Claude, Codex, and Gemini integrations" + } + ], + "methodology": "Endpoint stability analysis on Canva's production platform infrastructure", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Canva MCP Tools and Rate Limits Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "Per-tool rate limits are published in requests per minute and are tight on generation and editing endpoints (commonly around 20 req/min), constraining sustained agentic workloads" + } + ], + "methodology": "Rate limiting behavior observation against published per-tool limits under sustained load", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Tools Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "Transactional editing (start-editing-transaction, perform-editing-operations, commit-editing-transaction) isolates failed edits before commit; structured errors are returned for plan-gated tools and rate-limit hits" + } + ], + "methodology": "Error handling testing across plan-gated tools, invalid operations, and aborted editing transactions", + "last_verified": "2026-07-09" + }, + "large_asset_handling": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Tools Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "Asset upload is by URL (upload-asset-from-url) rather than inline payloads; multi-page design content retrieval (get-design-content, get-design-pages) can produce large tool results on complex designs" + } + ], + "methodology": "Testing content retrieval and export on large multi-page designs and heavy asset libraries", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 73, + "criteria": { + "authentication_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Canva MCP Documentation", + "url": "https://www.canva.dev/docs/mcp/", + "date": "2026-07-09", + "value": "Per-user OAuth is mandatory (every user needs their own Canva account); Client ID Metadata Documents (CIMD) are recommended so clients need no pre-registration or secrets, and custom clients must have redirect URIs approved onto an allowlist after review" + } + ], + "methodology": "Review of OAuth flow, CIMD support, redirect-URI allowlisting, and per-user authentication requirements", + "last_verified": "2026-07-09" + }, + "scope_limitation": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Canva Help Center: AI Connector Setup", + "url": "https://www.canva.com/help/mcp-agent-setup/", + "date": "2026-07-09", + "value": "Access mirrors the authenticated user's existing Canva permissions and plan entitlements (resize Pro+, autofill/brand kits Enterprise); there is no finer-grained read-only or per-folder scoping below the account level" + } + ], + "methodology": "Permission boundary testing across designs, folders, and brand assets the authenticated user can access", + "last_verified": "2026-07-09" + }, + "token_exposure_risk": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Documentation", + "url": "https://www.canva.dev/docs/mcp/", + "date": "2026-07-09", + "value": "OAuth tokens are managed by the MCP client rather than pasted as static API keys; CIMD eliminates client secrets entirely, and no local credential storage is required for supported clients" + } + ], + "methodology": "Token storage and exposure-surface analysis for the hosted OAuth transport", + "last_verified": "2026-07-09" + }, + "prompt_injection_risk": { + "score": 60, + "confidence": "medium", + "evidence": [ + { + "source": "Canva Help Center: MCP Actions", + "url": "https://www.canva.com/help/mcp-canva-usage/", + "date": "2026-07-09", + "value": "Shared designs, team comments, and imported content are third-party-authored input; text read from shared designs or comment threads can act as an injection vector into the consuming agent" + } + ], + "methodology": "Threat modeling of untrusted design and comment content flowing into agent context via read tools", + "last_verified": "2026-07-09" + }, + "unauthorized_action_risk": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Tools Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "Write tools can create and edit designs, move items between folders, post comments, upload assets, and (on Enterprise) autofill brand templates — modifying shared brand assets without a server-side confirmation step beyond client-level approvals" + } + ], + "methodology": "Authorization boundary testing of write-capable tools against shared designs and brand assets", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 72, + "criteria": { + "design_data_exposure": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "Canva Help Center: MCP Actions", + "url": "https://www.canva.com/help/mcp-canva-usage/", + "date": "2026-07-09", + "value": "Design content, text layers, comments, and brand assets returned by tools flow into the connected LLM provider's context; AI assistants can access what the user can view within existing permissions" + } + ], + "methodology": "Data flow analysis from Canva designs through MCP tool results to LLM providers", + "last_verified": "2026-07-09" + }, + "sensitive_data_protection": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Usage Policy", + "url": "https://www.canva.dev/docs/mcp/usage-policy/", + "date": "2026-07-09", + "value": "No built-in redaction of sensitive content embedded in designs (e.g., internal decks, financial figures in presentations, unreleased brand material); protection relies on the user connecting only appropriate accounts" + } + ], + "methodology": "Assessment of filtering and redaction controls on extracted design content", + "last_verified": "2026-07-09" + }, + "organization_data_control": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Canva AI Connector", + "url": "https://www.canva.com/ai-connector/", + "date": "2026-07-09", + "value": "Access is per-user and bounded by existing workspace permissions; Canva states enterprise-grade security and organizational policies apply, and Enterprise admins control brand kit and template access" + } + ], + "methodology": "Review of organizational access controls and admin policies applicable to MCP-connected accounts", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Canva Privacy Policy", + "url": "https://www.canva.com/policies/privacy-policy/", + "date": "2026-07-09", + "value": "Design data retrieved via MCP is shared with whichever LLM provider the user's client uses, per that provider's data policy rather than Canva's" + } + ], + "methodology": "Analysis of downstream data sharing once content leaves the Canva boundary", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 69, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Canva MCP Documentation", + "url": "https://www.canva.dev/docs/mcp/", + "date": "2026-07-09", + "value": "Dedicated developer documentation covering setup, the full tool catalog with per-tool rate limits, plan gating, usage policy, and Canva AI connector integration, plus Help Center guides for end users" + } + ], + "methodology": "Documentation completeness and accuracy review", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Canva Help Center: MCP Actions", + "url": "https://www.canva.com/help/mcp-canva-usage/", + "date": "2026-07-09", + "value": "Design edits appear in version history and comments are attributed to the authenticated user, but there is no dedicated MCP audit log distinguishing agent actions from manual ones" + } + ], + "methodology": "Logging and traceability assessment across client logs and Canva version history", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 35, + "confidence": "high", + "evidence": [ + { + "source": "Canva MCP Documentation", + "url": "https://www.canva.dev/docs/mcp/", + "date": "2026-07-09", + "value": "Server implementation is closed source and maintained by Canva; behavior can only be verified through documentation and observed tool output" + } + ], + "methodology": "Source availability and independent verifiability review", + "last_verified": "2026-07-09" + }, + "api_coverage_clarity": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Canva MCP Tools Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "Roughly 32 tools are individually enumerated with rate limits and plan requirements: design generation/lookup, editing transactions, assets, folders, exports and formats, comments and replies, import from URL, and resize" + } + ], + "methodology": "Comparison of documented tool surface against observed server capabilities", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 79, + "criteria": { + "ease_of_setup": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Canva Help Center: AI Connector Setup", + "url": "https://www.canva.com/help/mcp-agent-setup/", + "date": "2026-07-09", + "value": "Native connector flows exist for Claude and ChatGPT (add https://mcp.canva.com/mcp and complete OAuth); clients without remote MCP support use the mcp-remote shim, and custom clients must apply to register redirect URIs" + } + ], + "methodology": "Setup complexity assessment across first-party connectors, generic MCP clients, and custom clients", + "last_verified": "2026-07-09" + }, + "api_performance": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Canva MCP Tools Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "Design generation and export are heavyweight operations with latency scaling in design complexity and export format; tight per-minute limits cap parallel throughput" + } + ], + "methodology": "Latency observation across generation, editing, and export tool types", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Canva AI Connector", + "url": "https://www.canva.com/ai-connector/", + "date": "2026-07-09", + "value": "Generally available as a promoted first-party product on Canva's production platform with named integrations (ChatGPT, Claude, and others); tool surface is still expanding" + } + ], + "methodology": "Stability assessment of the hosted endpoint and change frequency of the tool surface", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Canva MCP Tools Documentation", + "url": "https://www.canva.dev/docs/mcp/tools/", + "date": "2026-07-09", + "value": "Covers the core design lifecycle end to end: generate, edit transactionally, organize in folders, comment, resize (Pro+), autofill brand templates (Enterprise), and export to PDF/PNG/JPG/PPTX/MP4" + } + ], + "methodology": "Feature completeness assessment against design-workflow needs across plan tiers", + "last_verified": "2026-07-09" + }, + "community_adoption": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Canva AI Connector", + "url": "https://www.canva.com/ai-connector/", + "date": "2026-07-09", + "value": "Promoted first-party integrations with ChatGPT, Claude, Codex, and Gemini, and third-party aggregators (Composio, StackOne, Zapier) ship Canva MCP connectors" + } + ], + "methodology": "Adoption analysis across MCP client ecosystems and integration platforms", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Per-user OAuth with CIMD support and a reviewed redirect-URI allowlist for custom clients", + "Transactional editing model isolates and validates changes before commit", + "Full design lifecycle coverage: generate, edit, organize, comment, resize, autofill, export", + "Per-tool rate limits and plan requirements are transparently published", + "Enterprise brand controls (brand kits, brand templates) integrate with existing admin governance", + "Native connector flows in major assistants (Claude, ChatGPT) with mcp-remote fallback elsewhere" + ], + "limitations": [ + "Closed source; server behavior cannot be independently audited", + "Write access extends to shared designs and (on Enterprise) brand assets with no server-side confirmation", + "Shared designs and comments are third-party-authored input and a prompt injection vector", + "Design content, including sensitive text in internal decks, is sent to the LLM provider", + "Key capabilities are plan-gated: resize requires Pro+, autofill and brand kits require Enterprise", + "Tight per-tool rate limits (around 20 req/min on generation/editing) constrain agentic throughput", + "No dedicated MCP audit log separating agent actions from manual edits" + ], + "metadata": { + "license": "Proprietary (closed source)", + "maintained_by": "Canva", + "status": "Generally available (AI Connector); custom-client redirect URI registration is application-based", + "remote_endpoint": "https://mcp.canva.com/mcp", + "authentication": "OAuth per user (CIMD recommended; no client secrets); custom clients register redirect URIs via an allowlist application", + "transport_types": [ + "streamable-http (hosted)" + ], + "installation_methods": [ + "Remote MCP endpoint", + "Native Claude/ChatGPT connector", + "mcp-remote shim for stdio-only clients" + ], + "pricing": "Included with Canva plans; feature gating by tier — resize requires Pro or above, brand-template autofill and brand kits require Enterprise (verified 2026-07-09)", + "tool_count": "~32 documented tools with published per-tool rate limits", + "mcp_version": "1.0" + }, + "use_case_ratings": { + "content-creation": { + "overall": 88, + "notes": "Primary use case: generating, editing, and exporting marketing and presentation designs from natural language" + }, + "creative-writing": { + "overall": 68, + "notes": "Useful for placing and revising copy within designs, though not a writing surface itself" + }, + "customer-support": { + "overall": 55, + "notes": "Marginal fit; limited to producing visual collateral for support content" + }, + "education": { + "overall": 78, + "notes": "Strong for producing teaching materials, worksheets, and presentation decks conversationally" + }, + "research-assistant": { + "overall": 58, + "notes": "Limited to searching the user's own design library and extracting design content" + } + }, + "best_for": [ + "Marketing teams generating on-brand collateral through AI assistants with Enterprise brand templates", + "Individual creators producing and exporting designs (PDF/PNG/PPTX/MP4) from chat workflows", + "Teams automating repetitive design production via brand-template autofill (Enterprise)", + "Assistants that need to search, organize, and comment on existing design libraries" + ], + "not_recommended_for": [ + "High-volume automated design pipelines (tight per-tool rate limits)", + "Organizations requiring auditable separation of agent versus human edits", + "Free-plan workflows needing resize or brand-kit capabilities" + ], + "related_entities": [ + "mcp-server-figma", + "mcp-server-notion", + "mcp-server-slack", + "mcp-server-shadcn" + ], + "tags": [ + "design", + "content-creation", + "canva", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-clickhouse.json b/data/mcps/mcp-server-clickhouse.json new file mode 100644 index 0000000..04968fa --- /dev/null +++ b/data/mcps/mcp-server-clickhouse.json @@ -0,0 +1,443 @@ +{ + "id": "mcp-server-clickhouse", + "type": "mcp", + "name": "MCP ClickHouse Server", + "provider": "ClickHouse (Official)", + "version": "mcp-clickhouse 0.4.0 (2026-06-03); Cloud Remote MCP; Managed ClickStack MCP", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Official ClickHouse MCP family: open-source mcp-clickhouse (PyPI, v0.4.0) plus ClickHouse Cloud Remote MCP at https://mcp.clickhouse.cloud/mcp with OAuth 2.0 and a Managed ClickStack MCP endpoint for observability. The local server is read-only by default; writes require CLICKHOUSE_ALLOW_WRITE_ACCESS=true, and destructive operations (DROP/TRUNCATE) need a second opt-in flag, CLICKHOUSE_ALLOW_DROP=true. Cloud remote tools are strictly read-only (readOnlyHint).", + "website": "https://github.com/ClickHouse/mcp-clickhouse", + "trust_vector": { + "performance_reliability": { + "overall_score": 83, + "criteria": { + "query_execution": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "run_query executes SQL on ClickHouse clusters via the official client, inheriting ClickHouse's columnar analytical performance; optional chDB tool provides embedded in-process querying" + } + ], + "methodology": "Query execution capability review", + "last_verified": "2026-07-09" + }, + "connection_stability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Built on the official ClickHouse Python client with configurable connection env vars; supports ClickHouse Cloud and self-managed clusters, with an unauthenticated /health endpoint on HTTP/SSE transports" + } + ], + "methodology": "Connection stability assessment", + "last_verified": "2026-07-09" + }, + "large_result_handling": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "list_tables supports pagination and filtering; very large SELECT result sets still need query-side LIMITs to avoid flooding the model context" + } + ], + "methodology": "Large result handling review", + "last_verified": "2026-07-09" + }, + "error_handling": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Middleware system allows request interception and custom handling; write/DDL attempts in read-only mode are rejected with explicit errors rather than executed" + } + ], + "methodology": "Error handling review", + "last_verified": "2026-07-09" + }, + "remote_endpoint_reliability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "ClickHouse Cloud Remote MCP documentation", + "url": "https://clickhouse.com/docs/cloud/features/ai-ml/remote-mcp", + "date": "2026-07-09", + "value": "Fully managed remote MCP server at https://mcp.clickhouse.cloud/mcp, enabled per service from the Cloud console Connect menu; 13 read-only tools covering querying, schema exploration, service management, backups, ClickPipes, and billing" + } + ], + "methodology": "Managed endpoint reliability review", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 78, + "criteria": { + "read_only_default": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Local server defaults to read-only, preventing accidental mutations during AI exploration; the Cloud remote server goes further — all tools carry readOnlyHint: true and no tool can modify data or service configuration" + } + ], + "methodology": "Default-posture review; secure-by-default read-only stance on both local and remote variants", + "last_verified": "2026-07-09" + }, + "graduated_write_gate": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Two-tier opt-in: CLICKHOUSE_ALLOW_WRITE_ACCESS=true enables DDL/DML (CREATE, ALTER, INSERT, UPDATE, DELETE), while destructive operations (DROP TABLE, DROP DATABASE, TRUNCATE) require a separate CLICKHOUSE_ALLOW_DROP=true flag" + } + ], + "methodology": "Write-gate design review; the graduated two-flag model separating writes from destructive DDL is a good practice other database MCP servers should adopt", + "last_verified": "2026-07-09" + }, + "authentication_security": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse Cloud Remote MCP documentation", + "url": "https://clickhouse.com/docs/cloud/features/ai-ml/remote-mcp", + "date": "2026-07-09", + "value": "Remote MCP authenticates via OAuth 2.0 scoped to the user's organizations and services; local HTTP/SSE transports require a bearer token (CLICKHOUSE_MCP_AUTH_TOKEN) or OAuth/OIDC delegation, with a development-only auth bypass flag" + } + ], + "methodology": "Authentication mechanism review; strong OAuth on remote, but local deployments can be misconfigured with the auth-disabled bypass", + "last_verified": "2026-07-09" + }, + "credential_exposure": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Local stdio deployments place CLICKHOUSE_USER/CLICKHOUSE_PASSWORD in environment variables or client config files; the OAuth-based Cloud remote avoids static credentials entirely" + } + ], + "methodology": "Credential security analysis; scored on the local path since it is the most common deployment", + "last_verified": "2026-07-09" + }, + "query_injection_risk": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Model Context Protocol security best practices", + "url": "https://modelcontextprotocol.io/specification/2025-06-18/basic/security_best_practices", + "date": "2026-07-09", + "value": "The AI constructs arbitrary SQL; untrusted content in query results can carry injected instructions back to the model — read-only default bounds the blast radius but exfiltration of readable data remains possible" + } + ], + "methodology": "Injection and prompt-injection exposure analysis", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 70, + "criteria": { + "data_exposure_to_llm": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "MCP Architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "Query results, schema listings, and (for ClickStack) telemetry data returned by tools are sent to the connected LLM provider" + } + ], + "methodology": "Data flow analysis", + "last_verified": "2026-07-09" + }, + "pii_protection": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "No built-in PII detection or result filtering; any data readable by the configured ClickHouse user can reach the model" + } + ], + "methodology": "PII protection assessment", + "last_verified": "2026-07-09" + }, + "self_hosted_option": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Full self-hosting: pip/uv install, Docker, stdio/HTTP/SSE transports, and chDB embedded mode allow entirely local operation against self-managed ClickHouse" + } + ], + "methodology": "Self-hosting options review", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "ClickHouse Cloud Remote MCP documentation", + "url": "https://clickhouse.com/docs/cloud/features/ai-ml/remote-mcp", + "date": "2026-07-09", + "value": "Remote MCP access is scoped to the authenticated user's organizations and services; database content shared with the LLM provider is governed by that provider's terms, not ClickHouse's" + } + ], + "methodology": "Data sharing analysis", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 82, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse Cloud Remote MCP documentation", + "url": "https://clickhouse.com/docs/cloud/features/ai-ml/remote-mcp", + "date": "2026-07-09", + "value": "First-party docs cover the repo README (env vars, flags, transports), Cloud remote MCP enablement, and ClickStack MCP blogs with setup for Claude Code, Cursor, and custom agents" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Apache 2.0, officially maintained under the ClickHouse GitHub org (818+ stars, 192 forks); v0.4.0 released 2026-06-03 on PyPI as mcp-clickhouse" + } + ], + "methodology": "Source code and maintenance review", + "last_verified": "2026-07-09" + }, + "query_visibility": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "ClickHouse query_log documentation", + "url": "https://clickhouse.com/docs/operations/system-tables/query_log", + "date": "2026-07-09", + "value": "All agent-issued SQL is recorded in ClickHouse's system.query_log; MCP clients additionally surface each tool call for user review" + } + ], + "methodology": "Query logging assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse blog — agentic analytics and ClickStack MCP", + "url": "https://clickhouse.com/blog/announcing-managed-clickstack-mcp-server", + "date": "2026-07-09", + "value": "Active first-party investment: remote MCP beta launch, AWS Marketplace AI Agents listing, Managed ClickStack MCP for observability agents, and regular releases on the open-source server" + } + ], + "methodology": "Community engagement analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 83, + "criteria": { + "ease_of_setup": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Local: uv/pip install mcp-clickhouse with a handful of env vars, plus an SQL Playground demo preset; Cloud: enable MCP from the service Connect menu and complete browser OAuth" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "query_performance": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse benchmarks", + "url": "https://benchmark.clickhouse.com/", + "date": "2026-07-09", + "value": "ClickHouse's columnar engine delivers sub-second analytical queries over billions of rows, making agent-driven exploratory analytics responsive" + } + ], + "methodology": "Performance review of the underlying engine for agent workloads", + "last_verified": "2026-07-09" + }, + "deployment_flexibility": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "Four deployment shapes: local stdio, self-hosted HTTP/SSE, managed Cloud Remote MCP (mcp.clickhouse.cloud/mcp), and Managed ClickStack MCP (mcp.clickhouse.cloud/clickstack) for observability, plus embedded chDB mode" + } + ], + "methodology": "Deployment options review", + "last_verified": "2026-07-09" + }, + "cost_efficiency": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "ClickHouse Cloud pricing", + "url": "https://clickhouse.com/pricing", + "date": "2026-07-09", + "value": "Open-source server and self-managed ClickHouse are free; Cloud remote MCP has no separate fee but agent queries consume service compute" + } + ], + "methodology": "Cost analysis", + "last_verified": "2026-07-09" + }, + "community_support": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "ClickHouse MCP server repository", + "url": "https://github.com/ClickHouse/mcp-clickhouse", + "date": "2026-07-09", + "value": "818 stars, 192 forks, active issue tracker and releases (v0.4.0, 2026-06-03), maintained by ClickHouse with community contributions" + } + ], + "methodology": "Community support assessment", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Read-only by default with a graduated write-gate: writes need CLICKHOUSE_ALLOW_WRITE_ACCESS=true and destructive DDL needs a second flag (CLICKHOUSE_ALLOW_DROP=true) — exemplary secure-by-default design", + "Managed Cloud Remote MCP with OAuth 2.0 and strictly read-only tools (readOnlyHint on all 13 tools)", + "Excellent analytical query performance for agent-driven data exploration", + "Flexible deployment: stdio, HTTP/SSE, managed remote, ClickStack observability endpoint, and embedded chDB", + "Open source (Apache 2.0) under the official ClickHouse org with active releases", + "Managed ClickStack MCP extends the same model to logs, metrics, and traces for incident-investigation agents" + ], + "limitations": [ + "Query results are exposed to the LLM provider with no built-in PII detection or filtering", + "Local stdio deployments store database credentials in environment variables or client config", + "Development-only auth bypass flag (CLICKHOUSE_MCP_AUTH_DISABLED) can be dangerously misused on exposed HTTP transports", + "Once both write flags are enabled, the agent can perform destructive DDL within the DB user's grants", + "Prompt-injection via untrusted data in query results remains possible; read-only mode limits damage but not data exfiltration", + "Large result sets require query-side LIMITs to avoid flooding model context", + "Cloud Remote MCP requires ClickHouse Cloud; feature is disabled per service until explicitly enabled" + ], + "metadata": { + "license": "Apache 2.0", + "supported_platforms": ["All platforms with Python 3.10+; Docker; any MCP client (Cloud remote)"], + "programming_languages": ["Python"], + "mcp_version": "1.0", + "github_repo": "https://github.com/ClickHouse/mcp-clickhouse", + "github_stars": 818, + "package_name": "mcp-clickhouse (PyPI)", + "latest_version": "0.4.0 (2026-06-03)", + "remote_endpoint": "https://mcp.clickhouse.cloud/mcp (Cloud Remote MCP); https://mcp.clickhouse.cloud/clickstack (Managed ClickStack MCP)", + "api_dependency": "ClickHouse Python client; chDB (optional embedded engine)", + "authentication": "Env-var credentials (stdio); bearer token or OAuth/OIDC (HTTP/SSE); OAuth 2.0 (Cloud remote)", + "security_controls": ["read-only default", "CLICKHOUSE_ALLOW_WRITE_ACCESS opt-in for DDL/DML", "CLICKHOUSE_ALLOW_DROP second opt-in for destructive operations", "readOnlyHint on all Cloud remote tools"], + "transport_types": ["stdio (default)", "http", "sse", "streamable-http (Cloud remote)"], + "maintained_by": "ClickHouse" + }, + "use_case_ratings": { + "code-generation": { + "overall": 80, + "notes": "Strong for generating and validating ClickHouse SQL against live schemas" + }, + "customer-support": { + "overall": 68, + "notes": "Useful for querying product analytics behind support questions" + }, + "content-creation": { + "overall": 60, + "notes": "Limited fit; data-backed reporting and dashboard narratives" + }, + "data-analysis": { + "overall": 94, + "notes": "Primary use case: fast agent-driven exploration of large analytical datasets" + }, + "research-assistant": { + "overall": 82, + "notes": "Great for iterative hypothesis testing over large datasets" + }, + "legal-compliance": { + "overall": 58, + "notes": "No PII filtering; requires scoped database users and read-only mode" + }, + "healthcare": { + "overall": 52, + "notes": "Telemetry/analytics fit, but PHI exposure to the LLM needs strict controls" + }, + "financial-analysis": { + "overall": 78, + "notes": "Excellent for market/event analytics; keep write flags off for production data" + }, + "education": { + "overall": 86, + "notes": "SQL Playground preset makes it easy to teach analytical SQL with agents" + }, + "creative-writing": { + "overall": 45, + "notes": "Minimal applicability beyond data lookups" + } + }, + "best_for": [ + "Analysts and engineers doing agent-driven exploration of large analytical datasets", + "Teams wanting secure-by-default database access with explicit, graduated write opt-ins", + "ClickHouse Cloud users preferring OAuth-scoped, strictly read-only managed access", + "Observability teams connecting incident-investigation agents via Managed ClickStack MCP" + ], + "related_entities": ["mcp-server-postgres", "mcp-server-elasticsearch", "mcp-server-snowflake", "mcp-server-databricks"], + "tags": ["database", "analytics", "olap", "observability", "sql", "mcp", "model-context-protocol", "clickhouse"] +} diff --git a/data/mcps/mcp-server-databricks.json b/data/mcps/mcp-server-databricks.json new file mode 100644 index 0000000..2ceea95 --- /dev/null +++ b/data/mcps/mcp-server-databricks.json @@ -0,0 +1,441 @@ +{ + "id": "mcp-server-databricks", + "type": "mcp", + "name": "MCP Databricks Managed Servers", + "provider": "Databricks (Official)", + "version": "Managed MCP servers (Genie Agent, AI Search, SQL, UC Functions GA; Genie One Beta)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Databricks managed MCP servers: workspace-hosted endpoints under https:///api/2.0/mcp/ for Genie natural-language data queries, AI Search (formerly vector-search; the legacy /api/2.0/mcp/vector-search prefix still works), Databricks SQL execution, and Unity Catalog functions as tools. OAuth with per-server scopes; Unity Catalog enforces permissions on every tool call and traffic is monitorable via the AI Gateway. Zero infrastructure — Databricks hosts and manages auth.", + "website": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 82, + "criteria": { + "api_reliability": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Workspace-hosted endpoints operated by Databricks; Genie Agent, AI Search, Databricks SQL, and Unity Catalog Functions servers are GA, with Genie One in Beta" + } + ], + "methodology": "Endpoint availability and GA-status review", + "last_verified": "2026-07-09" + }, + "genie_nl_accuracy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Genie Agent server answers natural-language questions over curated Genie spaces (read-only, asynchronous); accuracy depends on how well the Genie space is curated with instructions and trusted assets" + } + ], + "methodology": "NL-to-data capability review; Genie space curation is the dominant accuracy factor", + "last_verified": "2026-07-09" + }, + "retrieval_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Databricks unstructured retrieval tools documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/agent-framework/unstructured-retrieval-tools", + "date": "2026-07-09", + "value": "AI Search server (GA) queries AI Search indexes (formerly Vector Search) for relevant documents; requires Databricks-managed embeddings for MCP use" + } + ], + "methodology": "Retrieval capability assessment", + "last_verified": "2026-07-09" + }, + "sql_execution": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Databricks SQL server (GA) executes AI-generated SQL asynchronously with read and write capability at https:///api/2.0/mcp/sql; UC Functions server runs predefined SQL tools" + } + ], + "methodology": "SQL tool feature review", + "last_verified": "2026-07-09" + }, + "error_handling": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Databricks managed MCP meta parameters documentation", + "url": "https://docs.databricks.com/aws/en/agents/mcp/managed-mcp-meta-param", + "date": "2026-07-09", + "value": "Meta parameters allow tuning of managed server behavior; asynchronous Genie and SQL operations require clients to handle polling and long-running call semantics" + } + ], + "methodology": "Failure-mode and async-behavior review", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 82, + "criteria": { + "authentication_security": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "OAuth-based authentication managed by Databricks with per-server scopes (genie, ai-search, sql, unity-catalog); no static credentials in MCP client configuration" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Unity Catalog enforces permissions on every tool call across all managed servers — agents can only reach data and functions the authenticated principal is granted" + } + ], + "methodology": "Access control model review; per-call Unity Catalog enforcement is among the strongest governance models of evaluated MCP servers", + "last_verified": "2026-07-09" + }, + "data_modification_control": { + "score": 74, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Genie servers are read-only, but the Databricks SQL server has read and write capability and UC Functions can execute arbitrary predefined logic — writes are bounded only by Unity Catalog grants, with no separate MCP-level read-only switch documented for the SQL server" + } + ], + "methodology": "Write-surface assessment; scored below read-only-by-default peers because SQL write capability rides on catalog grants rather than an explicit MCP opt-in", + "last_verified": "2026-07-09" + }, + "prompt_injection_resilience": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Model Context Protocol security best practices", + "url": "https://modelcontextprotocol.io/specification/2025-06-18/basic/security_best_practices", + "date": "2026-07-09", + "value": "Unity Catalog scoping limits what an injected instruction can touch, but retrieved documents and query results still reach the LLM unfiltered; no Databricks-specific result-sanitization countermeasure is documented" + } + ], + "methodology": "Prompt-injection exposure analysis", + "last_verified": "2026-07-09" + }, + "credential_management": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Databricks hosts the servers and manages authentication; OAuth tokens are scoped per server type, avoiding long-lived PATs or connection strings in client config" + } + ], + "methodology": "Credential handling review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 81, + "criteria": { + "data_residency": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "MCP endpoints are hosted on the customer's own workspace hostname in the workspace's cloud region (AWS, Azure, GCP variants documented); tool execution stays within the workspace" + } + ], + "methodology": "Data residency review", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Databricks Trust Center", + "url": "https://www.databricks.com/trust", + "date": "2026-07-09", + "value": "Underlying platform holds SOC 2 Type II, ISO 27001, HIPAA, PCI DSS, and FedRAMP authorizations that apply to workspace-hosted MCP execution" + } + ], + "methodology": "Platform certification review", + "last_verified": "2026-07-09" + }, + "data_exposure_to_llm": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "MCP Architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "Genie answers, retrieved documents, and SQL results returned by tools are sent to the connected LLM provider per its data terms" + } + ], + "methodology": "Data flow analysis", + "last_verified": "2026-07-09" + }, + "governance_policy_enforcement": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Databricks Unity Catalog documentation", + "url": "https://docs.databricks.com/aws/en/data-governance/unity-catalog/", + "date": "2026-07-09", + "value": "Unity Catalog row filters, column masks, and grants apply to agent tool calls the same as any other access path, with centralized monitoring via the AI Gateway" + } + ], + "methodology": "Governance enforcement review", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 75, + "criteria": { + "documentation_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "First-party docs across AWS/Azure/GCP cover endpoints, OAuth scopes, GA/Beta status per server, meta parameters, and client setup examples" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "Announcing managed MCP servers blog", + "url": "https://www.databricks.com/blog/announcing-managed-mcp-servers-unity-catalog-and-mosaic-ai-integration", + "date": "2025-06-18", + "value": "Managed MCP servers are a proprietary hosted service; server implementation is not open source, though the MCP protocol itself is open" + } + ], + "methodology": "Source availability review; scored down for closed implementation despite open protocol", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Agent traffic is centrally monitorable via the Unity/AI Gateway, and SQL and Genie activity appears in standard workspace audit logs and query history" + } + ], + "methodology": "Audit trail assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Databricks Community — Managed MCP Servers", + "url": "https://community.databricks.com/t5/community-articles/databricks-mcp-servers-on-databricks/td-p/153690", + "date": "2026-07-09", + "value": "Active first-party engagement through Databricks community articles, blog posts, and the databricks-mcp library examples; no public issue tracker for the managed servers themselves" + } + ], + "methodology": "Community engagement analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 84, + "criteria": { + "ease_of_setup": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Databricks managed MCP servers documentation", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "date": "2026-07-09", + "value": "Zero setup: Databricks hosts the servers and manages authentication — clients add the workspace URL for the desired server (e.g. /api/2.0/mcp/genie/{space_id}) and complete the OAuth flow" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Databricks SQL warehouses documentation", + "url": "https://docs.databricks.com/aws/en/compute/sql-warehouse/", + "date": "2026-07-09", + "value": "SQL and Genie tool calls execute on serverless/SQL warehouse compute that scales elastically; asynchronous execution decouples MCP calls from long-running queries" + } + ], + "methodology": "Scalability review", + "last_verified": "2026-07-09" + }, + "cost_efficiency": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Databricks pricing", + "url": "https://www.databricks.com/product/pricing", + "date": "2026-07-09", + "value": "No separate charge for the MCP endpoints, but agent-driven Genie, AI Search, and SQL calls consume DBUs; unbounded exploratory agent traffic can accumulate compute cost" + } + ], + "methodology": "Cost analysis for agent-driven workloads", + "last_verified": "2026-07-09" + }, + "integration_ecosystem": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Databricks MCP overview", + "url": "https://docs.databricks.com/aws/en/generative-ai/mcp/", + "date": "2026-07-09", + "value": "Works with Claude, Cursor, and any MCP client; integrates with Mosaic AI Agent Framework, Agent Bricks, and supports custom MCP servers on Databricks Apps alongside the managed ones" + } + ], + "methodology": "Ecosystem integration review", + "last_verified": "2026-07-09" + }, + "managed_operations": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Announcing managed MCP servers blog", + "url": "https://www.databricks.com/blog/announcing-managed-mcp-servers-unity-catalog-and-mosaic-ai-integration", + "date": "2025-06-18", + "value": "Databricks operates, patches, and secures the MCP endpoints; customers manage only Unity Catalog grants and Genie space curation" + } + ], + "methodology": "Operational burden assessment", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Unity Catalog permission enforcement on every tool call — governance identical to any other access path", + "Zero infrastructure: Databricks hosts the endpoints and manages OAuth with per-server scopes", + "Full data-agent surface: Genie NL queries, AI Search retrieval, SQL execution, and UC function tools", + "Genie Agent, AI Search, SQL, and UC Functions servers are GA with multi-cloud docs (AWS, Azure, GCP)", + "Centralized agent-traffic monitoring via the AI Gateway plus standard workspace audit logs", + "Backward-compatible endpoint evolution (legacy vector-search prefix still works after AI Search rename)" + ], + "limitations": [ + "Databricks SQL server has read and write capability bounded only by Unity Catalog grants — no MCP-level read-only switch documented", + "Tool results (documents, Genie answers, SQL output) are exposed to the connected LLM provider", + "Closed-source managed implementation; behavior cannot be independently audited", + "Genie One server is still Beta; Genie answer quality depends heavily on space curation", + "AI Search via MCP requires Databricks-managed embeddings", + "Asynchronous Genie/SQL semantics add client-side complexity for long-running calls", + "Agent-driven compute (DBUs) can accumulate cost without warehouse guardrails" + ], + "metadata": { + "license": "Proprietary (managed service)", + "supported_platforms": ["Any MCP client (remote streamable HTTP); Databricks on AWS, Azure, GCP"], + "programming_languages": ["N/A (managed service)"], + "mcp_version": "1.0", + "docs": "https://docs.databricks.com/aws/en/generative-ai/mcp/managed-mcp", + "remote_endpoint": "https:///api/2.0/mcp/{genie|genie/{space_id}|ai-search/{catalog}/{schema}/{index}|sql|functions/{catalog}/{schema}/{function}} (legacy vector-search prefix still supported)", + "api_dependency": "Genie, AI Search (formerly Vector Search), Databricks SQL, Unity Catalog Functions", + "authentication": "OAuth managed by Databricks with per-server scopes (genie, ai-search, sql, unity-catalog)", + "security_controls": ["Unity Catalog per-call permission enforcement", "row filters and column masks", "AI Gateway monitoring", "read-only Genie servers"], + "server_status": "Genie Agent, AI Search, Databricks SQL, UC Functions: GA; Genie One: Beta", + "first_release": "2025-06-18 (announcement); GA rollout through late 2025/early 2026", + "maintained_by": "Databricks" + }, + "use_case_ratings": { + "code-generation": { + "overall": 76, + "notes": "Useful for generating SQL and UC function calls against governed schemas" + }, + "customer-support": { + "overall": 74, + "notes": "AI Search retrieval over support corpora with catalog-scoped access works well" + }, + "content-creation": { + "overall": 66, + "notes": "Indirect fit; retrieval-augmented drafting from governed document indexes" + }, + "data-analysis": { + "overall": 92, + "notes": "Core use case: Genie NL queries and SQL execution over lakehouse data" + }, + "research-assistant": { + "overall": 84, + "notes": "Strong document retrieval plus structured query access in one governed surface" + }, + "legal-compliance": { + "overall": 75, + "notes": "Unity Catalog lineage, grants, and audit logs support compliance review workflows" + }, + "healthcare": { + "overall": 70, + "notes": "HIPAA-capable platform, but tool results still flow to the LLM provider" + }, + "financial-analysis": { + "overall": 86, + "notes": "Governed lakehouse analytics with column masking and per-call enforcement" + }, + "education": { + "overall": 72, + "notes": "Good for teaching governed analytics; workspace prerequisites raise the entry bar" + }, + "creative-writing": { + "overall": 48, + "notes": "Minimal applicability beyond data-grounded reference lookups" + } + }, + "best_for": [ + "Lakehouse teams exposing Genie spaces and vector indexes to AI agents with zero new infrastructure", + "Enterprises requiring per-call Unity Catalog governance over every agent data access", + "Agent builders combining structured (SQL/Genie) and unstructured (AI Search) retrieval in one platform", + "Organizations standardizing agent traffic monitoring through an AI gateway" + ], + "related_entities": ["mcp-server-snowflake", "mcp-server-postgres", "mcp-server-elasticsearch", "mcp-server-mongodb"], + "tags": ["database", "lakehouse", "data-warehouse", "vector-search", "nl-to-sql", "mcp", "model-context-protocol", "databricks", "managed"] +} diff --git a/data/mcps/mcp-server-exa.json b/data/mcps/mcp-server-exa.json new file mode 100644 index 0000000..64238be --- /dev/null +++ b/data/mcps/mcp-server-exa.json @@ -0,0 +1,444 @@ +{ + "id": "mcp-server-exa", + "type": "mcp", + "name": "MCP Exa Server", + "provider": "Exa Labs", + "version": "3.2.1", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Exa's official MCP server (exa-labs/exa-mcp-server v3.2.1, MIT) exposing neural/semantic web search, code search, crawling/content fetch, company and people research, and deep research tools. Hosted at https://mcp.exa.ai/mcp with an unauthenticated free tier (3 QPS, 150 calls/day) or an Exa API key for full access; also runs locally via npx exa-mcp-server. Read-only, agent-optimized results with source URLs; Instant Search (Feb 2026) delivers sub-150ms latency.", + "website": "https://exa.ai/mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 86, + "criteria": { + "search_accuracy": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Exa - The World's Fastest Search API", + "url": "https://exa.ai/blog/fastest-search-api", + "date": "2026-07-09", + "value": "Proprietary end-to-end neural search stack using embeddings and transformers to understand query meaning, purpose-built for AI agents rather than repackaged keyword search" + } + ], + "methodology": "Search quality review of Exa's neural/semantic retrieval approach", + "last_verified": "2026-07-09" + }, + "response_latency": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Exa changelog - Introducing Exa Instant Search", + "url": "https://exa.ai/docs/changelog/instant-search-launch", + "date": "2026-02-05", + "value": "Instant Search (type=instant) launched 2026-02-05 delivering sub-150ms neural search latency for real-time agent workflows, voice AI, and coding agents" + } + ], + "methodology": "Latency review based on the Instant Search launch and published benchmarks", + "last_verified": "2026-07-09" + }, + "content_extraction": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "exa-mcp-server repository", + "url": "https://github.com/exa-labs/exa-mcp-server", + "date": "2026-07-09", + "value": "web_fetch_exa retrieves full page content; crawling and advanced-search variants provide filtered extraction for agents" + } + ], + "methodology": "Content extraction tool review", + "last_verified": "2026-07-09" + }, + "result_freshness": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Exa documentation", + "url": "https://docs.exa.ai/", + "date": "2026-07-09", + "value": "Continuously refreshed index with date filtering; live crawling options fetch current page content when the index is stale" + } + ], + "methodology": "Freshness assessment of index plus live-crawl fallback", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Exa changelog", + "url": "https://exa.ai/docs/changelog", + "date": "2026-07-09", + "value": "Unauthenticated hosted tier enforces 3 QPS and 150 calls/day with clear limit errors; adding an API key lifts limits to plan quotas" + } + ], + "methodology": "Rate limiting behavior review across anonymous and keyed tiers", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 76, + "criteria": { + "api_key_security": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "exa-mcp-server repository", + "url": "https://github.com/exa-labs/exa-mcp-server", + "date": "2026-07-09", + "value": "Local server takes EXA_API_KEY as an environment variable; hosted endpoint accepts the key via URL parameter or header — URL-embedded keys risk leaking through logs and client config sharing. Unauthenticated mode requires no credential at all" + } + ], + "methodology": "Credential handling review across local, hosted-keyed, and anonymous modes", + "last_verified": "2026-07-09" + }, + "prompt_injection_exposure": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "exa-mcp-server repository", + "url": "https://github.com/exa-labs/exa-mcp-server", + "date": "2026-07-09", + "value": "Search snippets and fetched page content from arbitrary websites flow directly into the LLM context, creating an indirect prompt-injection vector; no content sanitization is documented. Blast radius is limited because all tools are read-only" + } + ], + "methodology": "Threat modeling of untrusted web content ingestion via search and fetch tools", + "last_verified": "2026-07-09" + }, + "data_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Exa privacy policy", + "url": "https://exa.ai/privacy", + "date": "2026-07-09", + "value": "Search queries are processed on Exa's servers; hosted MCP usage is subject to Exa's privacy policy, with anonymous free-tier calls not tied to an account" + } + ], + "methodology": "Data handling review of query processing paths", + "last_verified": "2026-07-09" + }, + "source_verification": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Exa documentation", + "url": "https://docs.exa.ai/", + "date": "2026-07-09", + "value": "All results carry source URLs; content is retrieved from real indexed pages rather than generated, enabling downstream verification" + } + ], + "methodology": "Source attribution and verifiability testing", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 74, + "criteria": { + "query_privacy": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Exa privacy policy", + "url": "https://exa.ai/privacy", + "date": "2026-07-09", + "value": "Queries transit and may be retained by Exa for service operation and improvement; agent queries can embed sensitive project context" + } + ], + "methodology": "Privacy policy review", + "last_verified": "2026-07-09" + }, + "data_retention": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Exa privacy policy", + "url": "https://exa.ai/privacy", + "date": "2026-07-09", + "value": "Standard API-service retention terms; no zero-retention mode documented for MCP traffic" + } + ], + "methodology": "Data retention terms review", + "last_verified": "2026-07-09" + }, + "third_party_sharing": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Exa terms of service", + "url": "https://exa.ai/terms", + "date": "2026-07-09", + "value": "Query data is used for service operation; results additionally flow to the connected LLM provider per that provider's policy" + } + ], + "methodology": "Third-party sharing pathway analysis", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 86, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Exa MCP documentation", + "url": "https://docs.exa.ai/", + "date": "2026-07-09", + "value": "Clear docs for hosted and local setup, tool selection via ?tools= query parameters, per-client configuration guides, and bundled Claude Skills for research workflows" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "result_attribution": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Exa documentation", + "url": "https://docs.exa.ai/", + "date": "2026-07-09", + "value": "Every search and research result includes source URLs for verification" + } + ], + "methodology": "Attribution testing of result payloads", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "exa-labs/exa-mcp-server repository", + "url": "https://github.com/exa-labs/exa-mcp-server", + "date": "2026-07-09", + "value": "MIT-licensed server with ~4.7k stars and 409 commits; package.json at v3.2.1 — the MCP layer is fully inspectable while the search backend remains a proprietary hosted API" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-07-09" + }, + "pricing_transparency": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Exa pricing", + "url": "https://exa.ai/pricing", + "date": "2026-07-09", + "value": "Published per-request API pricing plus a documented free MCP tier (3 QPS / 150 calls/day unauthenticated; 1,000 requests/month with a free API key)" + } + ], + "methodology": "Pricing documentation review", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 88, + "criteria": { + "ease_of_setup": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Exa MCP page", + "url": "https://exa.ai/mcp", + "date": "2026-07-09", + "value": "Hosted endpoint https://mcp.exa.ai/mcp works with zero credentials on the free tier — arguably the lowest-friction MCP onboarding among search servers; local install is one npx exa-mcp-server command" + } + ], + "methodology": "Setup complexity assessment across hosted and local modes", + "last_verified": "2026-07-09" + }, + "api_reliability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Exa status page", + "url": "https://status.exa.ai/", + "date": "2026-07-09", + "value": "Hosted API with public status page and high-availability infrastructure serving production agent workloads" + } + ], + "methodology": "Reliability assessment of the hosted service", + "last_verified": "2026-07-09" + }, + "cost_efficiency": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Exa changelog - MCP free tier", + "url": "https://exa.ai/docs/changelog", + "date": "2026-07-09", + "value": "Free unauthenticated tier (150 calls/day) plus free-plan API credits make evaluation costless; paid usage is metered per request" + } + ], + "methodology": "Cost analysis of free and paid tiers", + "last_verified": "2026-07-09" + }, + "ai_optimization": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "exa-labs/exa-mcp-server repository", + "url": "https://github.com/exa-labs/exa-mcp-server", + "date": "2026-07-09", + "value": "Purpose-built for agents: semantic search tuned to LLM queries, configurable tool subsets to save context, deep-research tools, and Instant Search for latency-sensitive agent loops" + } + ], + "methodology": "AI integration and agent-fit assessment", + "last_verified": "2026-07-09" + }, + "maintenance_activity": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "exa-labs/exa-mcp-server repository", + "url": "https://github.com/exa-labs/exa-mcp-server", + "date": "2026-07-09", + "value": "Actively maintained by Exa Labs (409 commits, v3.2.1) alongside rapid platform iteration (Exa 2.0, Instant Search in Feb 2026)" + } + ], + "methodology": "Repository activity and release-cadence analysis", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 85, + "notes": "Code search and doc lookup tools are strong for finding APIs, examples, and repos" + }, + "customer-support": { + "overall": 75, + "notes": "Good for locating product docs and third-party resources" + }, + "content-creation": { + "overall": 87, + "notes": "Excellent semantic research and fact-gathering with cited sources" + }, + "data-analysis": { + "overall": 76, + "notes": "Useful for external data gathering, company and market research" + }, + "research-assistant": { + "overall": 94, + "notes": "Outstanding: neural search, crawling, and deep-research tools built for agents" + }, + "legal-compliance": { + "overall": 70, + "notes": "Useful for regulatory research; verify findings against primary sources" + }, + "healthcare": { + "overall": 66, + "notes": "Can surface medical literature; not suitable for clinical decisions" + }, + "financial-analysis": { + "overall": 78, + "notes": "Strong company research tools; confirm figures from primary sources" + }, + "education": { + "overall": 88, + "notes": "Great for learning and semantic exploration of topics" + }, + "creative-writing": { + "overall": 82, + "notes": "Semantic search surfaces inspiration and niche references well" + } + }, + "best_for": [ + "AI agents needing fast, meaning-aware web search (sub-150ms Instant Search)", + "Deep research and RAG pipelines requiring cited, extractable web content", + "Developers searching code, repos, and technical documentation", + "Zero-friction evaluation via the unauthenticated hosted free tier" + ], + "strengths": [ + "Neural/semantic search purpose-built for AI agents, not repackaged keyword search", + "Instant Search (launched 2026-02-05) delivers sub-150ms latency for real-time agent loops", + "Unauthenticated hosted free tier (3 QPS / 150 calls/day) — lowest-friction onboarding in its class", + "Broad tool set: web search, fetch/crawl, code search, company and people research, deep research", + "Read-only surface keeps worst-case blast radius low", + "MIT open-source server (~4.7k stars, v3.2.1) with configurable tool subsets to save context", + "All results carry source URLs for verification" + ], + "limitations": [ + "Fetched web content flows into the LLM context — indirect prompt-injection vector with no documented sanitization", + "Hosted endpoint accepts API keys via URL parameter, which can leak through logs and shared configs", + "Queries processed and retained by Exa's servers per its privacy policy; no self-hosted backend", + "Anonymous tier is tightly rate-limited (150 calls/day), requiring a key for real workloads", + "Search backend is proprietary despite the open-source MCP layer", + "Result quality varies with source websites, as with any web search" + ], + "metadata": { + "license": "MIT (MCP server); proprietary hosted search API", + "supported_platforms": [ + "Any MCP client (hosted endpoint)", + "All platforms with Node.js (local server)" + ], + "programming_languages": [ + "TypeScript" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/exa-labs/exa-mcp-server", + "github_stars": 4700, + "package": "exa-mcp-server", + "package_version": "3.2.1", + "remote_endpoint": "https://mcp.exa.ai/mcp (tool selection via ?tools= query parameter)", + "api_dependency": "Exa Search API", + "authentication": "None required (free tier: 3 QPS / 150 calls/day); Exa API key (URL parameter or header) for full access", + "free_tier": "Unauthenticated: 3 QPS, 150 calls/day; free API key: 1,000 requests/month", + "first_release": "2024", + "maintained_by": "Exa Labs", + "status": "Active", + "transport_types": [ + "streamable-http (hosted)", + "stdio (local)" + ], + "installation_methods": [ + "remote-url", + "npx", + "npm" + ] + }, + "related_entities": [ + "mcp-server-tavily", + "mcp-server-brave-search", + "mcp-server-perplexity", + "mcp-server-firecrawl" + ], + "tags": [ + "search", + "web", + "semantic-search", + "research", + "exa", + "mcp", + "model-context-protocol", + "official" + ] +} diff --git a/data/mcps/mcp-server-grafana.json b/data/mcps/mcp-server-grafana.json new file mode 100644 index 0000000..bbdaf98 --- /dev/null +++ b/data/mcps/mcp-server-grafana.json @@ -0,0 +1,488 @@ +{ + "id": "mcp-server-grafana", + "type": "mcp", + "name": "MCP Grafana Server", + "provider": "Grafana Labs", + "version": "0.17.1", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Official Grafana Labs MCP server (grafana/mcp-grafana, Apache-2.0, Go) with 40+ tools: query metrics/logs/traces across Prometheus, Loki, and many other datasources, search and update dashboards, manage alert rules, and drive Incident, Sift, and OnCall. Supports stdio, SSE, and streamable HTTP with service-account token auth, category-level tool enablement, and a --disable-write read-only mode; also powers Grafana Assistant, with a hosted Grafana Cloud MCP endpoint in preview.", + "website": "https://github.com/grafana/mcp-grafana", + "trust_vector": { + "performance_reliability": { + "overall_score": 87, + "criteria": { + "metrics_query_reliability": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - Prometheus tools", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Dedicated Prometheus toolset (query_prometheus, list_prometheus_metric_metadata, label names/values) built on Grafana's mature datasource proxy APIs; requires Grafana 9.0+ for full functionality" + } + ], + "methodology": "Review of PromQL query tools and underlying Grafana datasource API maturity", + "last_verified": "2026-07-09" + }, + "log_search_accuracy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - Loki tools", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "LogQL query tools (query_loki_logs, query_loki_stats, label APIs) expose Grafana's Loki log search with stats endpoints to gauge result volume before querying" + } + ], + "methodology": "Log query tool review against Loki API capabilities", + "last_verified": "2026-07-09" + }, + "datasource_coverage": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - toolsets", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Tool categories span Prometheus, Loki, InfluxDB, ClickHouse, CloudWatch, Elasticsearch/OpenSearch, Graphite, Athena, Snowflake, Quickwit, and Pyroscope profiling, plus dashboards, alerting, incidents, Sift, and OnCall" + } + ], + "methodology": "Feature inventory of datasource-specific toolsets in the repository", + "last_verified": "2026-07-09" + }, + "dashboard_management_reliability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "mcp-grafana README - dashboard tools", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Dashboard search, get, update, and panel-level tools with JSON-patch style updates; write operations reuse Grafana's versioned dashboard API so failed saves do not corrupt existing dashboards" + } + ], + "methodology": "Review of dashboard read/write tool design and Grafana API versioning behavior", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "mcp-grafana repository", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Errors from Grafana APIs are returned as structured tool errors; slow-request diagnostics, Prometheus metrics, and OpenTelemetry tracing help operators diagnose failures, though retry behavior is left to the client" + } + ], + "methodology": "Error-path and observability feature review", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 73, + "criteria": { + "authentication_security": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - authentication", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Authenticates with a Grafana service-account token (GRAFANA_SERVICE_ACCOUNT_TOKEN); tokens can be read from files for rotation in Kubernetes; TLS client/server certificates and host/origin validation supported for HTTP transports" + }, + { + "source": "Grafana Cloud MCP server docs", + "url": "https://grafana.com/docs/grafana-cloud/machine-learning/assistant/configure/cloud-mcp/", + "date": "2026-07-09", + "value": "Hosted Grafana Cloud MCP server (public preview) uses OAuth 2.1 over streamable HTTP, removing long-lived tokens for cloud users" + } + ], + "methodology": "Authentication mechanism review across self-hosted and Grafana Cloud modes", + "last_verified": "2026-07-09" + }, + "write_action_control": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - configuration flags", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Mixed read/write surface (dashboards, alert rules, incidents, annotations, snapshots are writable) but a --disable-write flag provides read-only mode, and --enabled-tools / --disable- flags allow fine-grained allowlisting; admin tools disabled by default" + } + ], + "methodology": "Capability analysis of write-capable tools and available restriction flags", + "last_verified": "2026-07-09" + }, + "credential_exposure_risk": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "mcp-grafana README - setup", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Default local setup places a long-lived service-account token in client configuration or environment variables; scope depends on the roles granted to the service account rather than per-tool permissions" + } + ], + "methodology": "Credential storage and scoping analysis for the stdio/local deployment model", + "last_verified": "2026-07-09" + }, + "infrastructure_visibility_risk": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Security analysis of observability MCP access", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "An agent with datasource, dashboard, and alerting access can map infrastructure topology, service names, and alert thresholds; observability data frequently contains internal hostnames, IPs, and occasionally secrets leaked into logs" + } + ], + "methodology": "Visibility risk assessment of aggregated observability access", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "mcp-grafana README - observability", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Server exposes Prometheus metrics and OpenTelemetry tracing/logging (OTLP) for MCP operations; actions performed via the Grafana API are attributable to the service account, with full audit logs available in Grafana Enterprise/Cloud" + } + ], + "methodology": "Audit and traceability review of server-side and Grafana-side logging", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 72, + "criteria": { + "observability_data_exposure": { + "score": 65, + "confidence": "high", + "evidence": [ + { + "source": "MCP data flow architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "Metric series, log lines, trace spans, and dashboard definitions returned by tools are sent to the LLM provider as tool results" + } + ], + "methodology": "Data flow analysis of query tool outputs", + "last_verified": "2026-07-09" + }, + "log_data_privacy": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "Privacy analysis of Loki/Elasticsearch log access", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Application logs queried via Loki or Elasticsearch tools may contain PII, tokens, API keys, and other secrets accidentally logged by applications; the server performs no redaction" + } + ], + "methodology": "Log privacy assessment; no built-in scrubbing identified", + "last_verified": "2026-07-09" + }, + "self_hosted_data_control": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - deployment", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Server runs locally or in-cluster (binary, Docker, Helm chart) against self-hosted Grafana; no data passes through Grafana Labs infrastructure unless using Grafana Cloud" + } + ], + "methodology": "Deployment model and data residency review", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "MCP client documentation", + "url": "https://modelcontextprotocol.io/docs", + "date": "2026-07-09", + "value": "Observability data is shared only with the connected LLM provider per that provider's privacy policy; the open-source server itself sends no telemetry to Grafana Labs" + } + ], + "methodology": "Data sharing pathway analysis", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 90, + "criteria": { + "documentation_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README and Grafana docs", + "url": "https://grafana.com/docs/grafana/latest/developer-resources/mcp/", + "date": "2026-07-09", + "value": "Full tool reference, per-category enablement flags, transport configuration, RBAC permission documentation, and client setup guides maintained in both the repo and official Grafana documentation" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "grafana/mcp-grafana repository", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Fully open source under Apache-2.0 with ~3.2k stars and 397 forks; public issue tracker and release notes; v0.17.1 released 2026-07-07" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - observability features", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Every action is an explicit named tool call; server ships Prometheus metrics, OpenTelemetry traces/logs, health checks, and slow-request diagnostics for inspecting agent behavior" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-07-09" + }, + "vendor_credibility": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Grafana Labs", + "url": "https://grafana.com/docs/grafana/latest/developer-resources/mcp/", + "date": "2026-07-09", + "value": "Officially maintained by Grafana Labs and embedded as the tool layer of Grafana Assistant in Grafana Cloud; frequent releases through July 2026" + } + ], + "methodology": "Maintainer reputation and project health analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 88, + "criteria": { + "ease_of_setup": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - installation", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Install via release binary, Docker image, uvx, Go build, or Kubernetes Helm chart; requires creating a Grafana service account and token, adding some friction versus hosted OAuth servers" + } + ], + "methodology": "Setup complexity assessment across deployment options", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana tool reference", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "40+ tools covering dashboards, datasource queries (Prometheus, Loki, InfluxDB, ClickHouse, CloudWatch, Elasticsearch, Graphite, Athena, Snowflake, Quickwit, Pyroscope), alerting, Incident, Sift investigations, OnCall, annotations, snapshots, rendering, and deeplinks" + } + ], + "methodology": "Feature completeness assessment against observability workflow needs", + "last_verified": "2026-07-09" + }, + "maintenance_activity": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana releases", + "url": "https://github.com/grafana/mcp-grafana/releases", + "date": "2026-07-09", + "value": "v0.17.1 released 2026-07-07 with a rapid release cadence; active issue triage by the Grafana Labs team" + } + ], + "methodology": "Release cadence and commit activity analysis", + "last_verified": "2026-07-09" + }, + "integration_flexibility": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "mcp-grafana README - transports", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Three transports (stdio, SSE, streamable HTTP), Helm chart for in-cluster deployment, category-selectable tool enablement to control context size, and Grafana Cloud embedding via Grafana Assistant" + } + ], + "methodology": "Deployment and integration option review", + "last_verified": "2026-07-09" + }, + "performance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "mcp-grafana repository", + "url": "https://github.com/grafana/mcp-grafana", + "date": "2026-07-09", + "value": "Go implementation with low overhead; response times dominated by underlying Grafana/datasource query latency; tool-category selection keeps context windows manageable" + } + ], + "methodology": "Implementation and latency characteristics review", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 80, + "notes": "Good for generating dashboards-as-code, alert rules, and PromQL/LogQL queries" + }, + "customer-support": { + "overall": 75, + "notes": "Useful for investigating customer-reported latency or error spikes via metrics and logs" + }, + "content-creation": { + "overall": 50, + "notes": "Limited applicability beyond dashboard and report generation" + }, + "data-analysis": { + "overall": 93, + "notes": "Excellent for cross-datasource observability analytics, trend detection, and incident forensics" + }, + "research-assistant": { + "overall": 74, + "notes": "Useful for researching system behavior and performance patterns" + }, + "legal-compliance": { + "overall": 58, + "notes": "Risk of exposing infrastructure details and unscrubbed log data to the LLM" + }, + "healthcare": { + "overall": 55, + "notes": "PHI may leak through application logs and traces; requires strict log hygiene" + }, + "financial-analysis": { + "overall": 65, + "notes": "Moderate risk when monitoring financial infrastructure; self-hosting helps" + }, + "education": { + "overall": 85, + "notes": "Excellent for teaching observability, PromQL, LogQL, and alerting practices" + }, + "creative-writing": { + "overall": 40, + "notes": "Low relevance to creative writing workflows" + } + }, + "best_for": [ + "SREs investigating incidents across metrics, logs, traces, and profiles", + "DevOps teams managing dashboards and alert rules with AI assistance", + "Organizations wanting a self-hosted, open-source observability MCP with no vendor data path", + "Grafana Cloud users leveraging Grafana Assistant and the hosted MCP endpoint" + ], + "strengths": [ + "Official Grafana Labs server, Apache-2.0, ~3.2k stars, rapid releases (v0.17.1 on 2026-07-07)", + "40+ tools spanning Prometheus, Loki, and 9 more datasource types plus Incident, Sift, and OnCall", + "Category-selectable tool enablement and --disable-write read-only mode limit blast radius", + "Three transports (stdio, SSE, streamable HTTP) with TLS, Helm chart, and Kubernetes token rotation", + "Self-hostable end to end: observability data need never transit vendor infrastructure", + "Strong server-side observability: Prometheus metrics, OpenTelemetry traces, health checks", + "Also embedded in Grafana Cloud as the tool layer for Grafana Assistant (OAuth 2.1 hosted preview)" + ], + "limitations": [ + "Metric, log, and trace data flows into the LLM context and may contain secrets or PII", + "Default write access to dashboards, alert rules, and incidents unless --disable-write is set", + "Long-lived service-account token in client config for local setups (no OAuth outside Grafana Cloud)", + "Agent can map internal infrastructure topology, hostnames, and alert thresholds", + "No built-in redaction of sensitive values returned by log and trace queries", + "Requires Grafana 9.0+; some datasource tools unavailable on older versions", + "40+ tools can crowd the context window unless categories are pruned" + ], + "metadata": { + "license": "Apache-2.0", + "supported_platforms": [ + "macOS, Linux, Windows (binary)", + "Docker", + "Kubernetes (Helm chart)" + ], + "programming_languages": [ + "Go" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/grafana/mcp-grafana", + "github_stars": 3200, + "package_version": "0.17.1", + "release_date": "2026-07-07", + "api_dependency": "Grafana HTTP API (Grafana 9.0+)", + "authentication": "Grafana service-account token (env var or file); OAuth 2.1 on hosted Grafana Cloud MCP (public preview)", + "remote_endpoint": "https://mcp.grafana.com/mcp (Grafana Cloud, public preview)", + "security_controls": [ + "--disable-write read-only mode", + "--enabled-tools allowlist and per-category disable flags", + "TLS client/server certificates", + "host/origin validation for HTTP transports" + ], + "first_release": "2025", + "maintained_by": "Grafana Labs", + "status": "Active", + "transport_types": [ + "stdio", + "sse", + "streamable-http" + ], + "installation_methods": [ + "binary", + "docker", + "uvx", + "go install", + "helm" + ] + }, + "related_entities": [ + "mcp-server-datadog", + "mcp-server-sentry", + "mcp-server-elasticsearch", + "mcp-server-kubernetes" + ], + "tags": [ + "monitoring", + "observability", + "grafana", + "prometheus", + "loki", + "mcp", + "model-context-protocol", + "official" + ] +} diff --git a/data/mcps/mcp-server-hubspot.json b/data/mcps/mcp-server-hubspot.json new file mode 100644 index 0000000..ef3d1c7 --- /dev/null +++ b/data/mcps/mcp-server-hubspot.json @@ -0,0 +1,472 @@ +{ + "id": "mcp-server-hubspot", + "type": "mcp", + "name": "MCP HubSpot Server", + "provider": "HubSpot (Official)", + "version": "hosted remote (GA 2026-04-13)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "HubSpot's official hosted MCP server at https://mcp.hubspot.com, authenticated via OAuth 2.1 with PKCE and available to all HubSpot accounts and tiers. Launched as a read-only public beta in September 2025; GA on 2026-04-13 added write capabilities (CRM records and activities), activity history, marketing content objects, and organizational context - a material expansion of the risk surface. Actions inherit the connecting user's HubSpot permissions.", + "website": "https://developers.hubspot.com/ai-tools/mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 81, + "criteria": { + "crm_operation_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot MCP Server Documentation", + "url": "https://developers.hubspot.com/ai-tools/mcp", + "date": "2026-07-09", + "value": "Hosted by HubSpot on the same platform as its public CRM APIs; reliable record operations across contacts, companies, deals, tickets, line items, and products" + } + ], + "methodology": "Operation reliability assessment against the underlying HubSpot CRM API", + "last_verified": "2026-07-09" + }, + "search_accuracy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot CRM Search API", + "url": "https://developers.hubspot.com/docs/api/crm/search", + "date": "2026-07-09", + "value": "Record retrieval builds on the CRM search API with filtering across object types" + } + ], + "methodology": "Search quality testing", + "last_verified": "2026-07-09" + }, + "write_operation_reliability": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "GA added write access for CRM records (contacts, companies, deals, tickets, line items, products) and activities (calls, emails, meetings, notes, tasks) with validation via the CRM APIs" + } + ], + "methodology": "Write-path reliability review", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot API Usage Guidelines", + "url": "https://developers.hubspot.com/docs/api/usage-details", + "date": "2026-07-09", + "value": "Subject to HubSpot API rate limits, which vary by subscription tier" + } + ], + "methodology": "Rate limiting behavior analysis", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot MCP Server Documentation", + "url": "https://developers.hubspot.com/docs/apps/developer-platform/build-apps/integrate-with-the-remote-hubspot-mcp-server", + "date": "2026-07-09", + "value": "Standard API errors (permissions, validation, sensitive-data blocks) are surfaced to the MCP client; get_user_details tool lets agents discover available objects before acting" + } + ], + "methodology": "Error handling and discovery testing", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 73, + "criteria": { + "authentication_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "OAuth 2.1 with PKCE is required for all connections; credentials generated via an MCP auth app in the account's Development section - no static API keys" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-07-09" + }, + "scope_limitation": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "Actions respect existing HubSpot user permissions - the connection inherits the full scope of the authorizing user rather than offering per-connection narrowing (e.g. a read-only grant)" + } + ], + "methodology": "Analysis of available mechanisms to scope agent access below the user's permission level", + "last_verified": "2026-07-09" + }, + "write_action_risk": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "GA on 2026-04-13 changed the server from read-only (beta since Sept 2025) to read/write: agents can now create and update contacts, companies, deals, tickets, line items, products, and log calls, emails, meetings, notes, and tasks - a materially larger risk surface than the beta" + } + ], + "methodology": "Assessment of the GA capability change from read-only to write-enabled and its blast radius", + "last_verified": "2026-07-09" + }, + "sensitive_data_controls": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "Accounts with the sensitive-data setting enabled have Activity objects (calls, emails, meetings, notes, tasks) blocked from MCP access entirely" + } + ], + "methodology": "Review of built-in sensitive data safeguards", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot Account Activity Logging", + "url": "https://knowledge.hubspot.com/account-management/view-account-activity-history", + "date": "2026-07-09", + "value": "Record changes appear in HubSpot property/activity history attributed to the authorizing user; no dedicated MCP-level audit log distinguishing agent actions from user actions" + } + ], + "methodology": "Audit logging review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 66, + "criteria": { + "crm_pii_exposure": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "MCP Data Flow", + "url": "https://modelcontextprotocol.io/docs/architecture", + "date": "2026-07-09", + "value": "Contact and company records - names, emails, phone numbers, deal values - are returned into the agent context and sent to the LLM provider" + } + ], + "methodology": "Data flow analysis of CRM PII exposure", + "last_verified": "2026-07-09" + }, + "activity_and_communication_exposure": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "GA added activity history access: logged calls, emails, meetings, notes, and tasks - customer communication content - become readable by connected agents unless the sensitive-data block applies" + } + ], + "methodology": "Communication content exposure assessment", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "LLM Provider Policies", + "url": "https://modelcontextprotocol.io/docs", + "date": "2026-07-09", + "value": "CRM and marketing data retrieved via MCP is processed by the connected LLM provider under that provider's privacy policy" + } + ], + "methodology": "Data sharing analysis", + "last_verified": "2026-07-09" + }, + "tenant_scoping": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot MCP Server Documentation", + "url": "https://developers.hubspot.com/docs/apps/developer-platform/build-apps/integrate-with-the-remote-hubspot-mcp-server", + "date": "2026-07-09", + "value": "Each OAuth connection is bound to a specific HubSpot account and user; access is confined to records that user can view or edit" + } + ], + "methodology": "Tenant and account isolation review", + "last_verified": "2026-07-09" + }, + "compliance_readiness": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "Sensitive-data accounts get automatic activity blocking, and permissions are enforced, but GDPR/CCPA obligations for routing customer PII through third-party LLMs remain the customer's responsibility" + } + ], + "methodology": "Regulatory readiness assessment", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 78, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot MCP Server Documentation", + "url": "https://developers.hubspot.com/ai-tools/mcp", + "date": "2026-07-09", + "value": "First-party docs cover setup, OAuth app creation, supported objects for read vs write, and client integration guides" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "changelog_and_disclosure": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "The read-only-to-read/write capability change at GA was clearly announced in the developer changelog with an effective date and object-by-object scope" + } + ], + "methodology": "Review of capability-change communication", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot MCP Server Documentation", + "url": "https://developers.hubspot.com/docs/apps/developer-platform/build-apps/integrate-with-the-remote-hubspot-mcp-server", + "date": "2026-07-09", + "value": "Changes are visible in record property history, but agent-initiated actions are not separately labeled from the authorizing user's own actions" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot MCP Server Documentation", + "url": "https://developers.hubspot.com/ai-tools/mcp", + "date": "2026-07-09", + "value": "The remote server is a closed-source hosted service; behavior is documented but the implementation is not auditable" + } + ], + "methodology": "Source availability review", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 82, + "criteria": { + "ease_of_setup": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot MCP Server Documentation", + "url": "https://developers.hubspot.com/docs/apps/developer-platform/build-apps/integrate-with-the-remote-hubspot-mcp-server", + "date": "2026-07-09", + "value": "No install: create an MCP auth app in the account's Development section, add https://mcp.hubspot.com to the client, complete browser OAuth" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "api_performance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "HubSpot Developer Platform", + "url": "https://developers.hubspot.com/", + "date": "2026-07-09", + "value": "Performance tracks the underlying HubSpot CRM APIs; typical record operations complete quickly, large list operations are paginated" + } + ], + "methodology": "Performance assessment", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Status", + "url": "https://status.hubspot.com/", + "date": "2026-07-09", + "value": "Runs on HubSpot's production platform with public status transparency and strong uptime history" + } + ], + "methodology": "Reliability analysis", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "Reads span CRM records, marketing content (campaigns, landing pages, blog posts, site pages, marketing events), and activities; writes cover core CRM records and activities - but custom objects are not supported" + } + ], + "methodology": "Feature coverage assessment", + "last_verified": "2026-07-09" + }, + "availability_across_tiers": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "HubSpot Changelog - Remote MCP Server GA", + "url": "https://developers.hubspot.com/changelog/remote-hubspot-mcp-server-is-now-generally-available", + "date": "2026-04-13", + "value": "GA to all HubSpot accounts - every hub and tier - at no additional cost" + } + ], + "methodology": "Tier and hub availability review", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Official HubSpot-hosted remote at https://mcp.hubspot.com - no install, OAuth 2.1 with PKCE required, no static API keys", + "Available to all HubSpot accounts, hubs, and tiers at no additional cost", + "Broad coverage: CRM records, activity history, and marketing content objects, with read/write on core objects since GA", + "Inherits and enforces existing HubSpot user permissions on every action", + "Built-in safeguard: sensitive-data accounts have Activity objects blocked from MCP access automatically", + "Capability changes clearly communicated via the developer changelog (GA write expansion announced with scope and date)" + ], + "limitations": [ + "RISK-SURFACE CHANGE 2026-04-13: GA added write capabilities - agents can now create/update CRM records and log activities, where the Sept 2025 beta was read-only; connections authorized pre-GA gained write reach with the platform change", + "No per-connection scope narrowing: the agent gets the authorizing user's full permission set (no read-only grant option)", + "Contact/company PII and customer communication content (calls, emails, notes) flow into the LLM provider's context", + "Custom objects are not supported", + "No dedicated MCP audit log; agent actions are attributed to the authorizing user in record history", + "Closed-source hosted service; implementation not auditable", + "Subject to HubSpot API rate limits that vary by subscription tier" + ], + "metadata": { + "license": "Proprietary (hosted service)", + "supported_platforms": [ + "Hosted remote (https://mcp.hubspot.com)", + "Any MCP client with HTTP transport and OAuth 2.1 support" + ], + "programming_languages": [ + "N/A (hosted service)" + ], + "mcp_version": "1.0", + "docs": "https://developers.hubspot.com/ai-tools/mcp", + "api_dependency": "HubSpot CRM and Marketing APIs", + "authentication": "OAuth 2.1 with PKCE (required); MCP auth app created in the account's Development section", + "remote_endpoint": "https://mcp.hubspot.com", + "first_release": "2025-09 (public beta, read-only); 2026-04-13 (GA with write)", + "maintained_by": "HubSpot", + "status": "Active - generally available to all accounts", + "transport_types": [ + "streamable-http" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 60, + "notes": "Limited relevance; useful for generating CRM integration code against live schemas" + }, + "customer-support": { + "overall": 86, + "notes": "Strong fit: ticket, contact, and activity context with write-back for notes and tasks" + }, + "content-creation": { + "overall": 80, + "notes": "Marketing content objects (campaigns, landing pages, blogs) enable CRM-grounded content workflows" + }, + "data-analysis": { + "overall": 78, + "notes": "Good for pipeline and deal analysis; rate limits constrain large extractions" + }, + "research-assistant": { + "overall": 70, + "notes": "Useful for account and prospect research within the CRM" + }, + "legal-compliance": { + "overall": 56, + "notes": "Customer PII and communication content reach the LLM provider; sensitive-data block helps but is coarse" + }, + "healthcare": { + "overall": 52, + "notes": "Patient contact data in CRM routed through agent contexts is high risk" + }, + "financial-analysis": { + "overall": 66, + "notes": "Deal and revenue data accessible; moderate exposure risk" + }, + "education": { + "overall": 68, + "notes": "Workable for admissions/CRM use with standard PII caveats" + }, + "creative-writing": { + "overall": 45, + "notes": "Marginal fit beyond marketing copy grounded in CRM data" + } + }, + "best_for": [ + "Sales and marketing teams driving CRM updates from AI assistants", + "Support workflows that read ticket/contact context and write back notes and tasks", + "RevOps teams analyzing pipelines conversationally across all HubSpot tiers" + ], + "related_entities": [ + "mcp-server-stripe", + "mcp-server-zapier", + "mcp-server-slack" + ], + "tags": [ + "crm", + "marketing", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ] +} diff --git a/data/mcps/mcp-server-neo4j.json b/data/mcps/mcp-server-neo4j.json new file mode 100644 index 0000000..52bea82 --- /dev/null +++ b/data/mcps/mcp-server-neo4j.json @@ -0,0 +1,445 @@ +{ + "id": "mcp-server-neo4j", + "type": "mcp", + "name": "MCP Neo4j Server", + "provider": "Neo4j (Official)", + "version": "1.5.3 (2026-06-11)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Official Neo4j MCP server (neo4j/mcp; neo4j-mcp-server on PyPI, v1.5.3 released 2026-06-11): schema introspection, read-cypher (read-only enforced via EXPLAIN query classification), write-cypher, and GDS procedure listing over stdio. Arbitrary Cypher means read+write access to the graph unless NEO4J_READ_ONLY=true is set. Distinct from the experimental Neo4j Labs servers (mcp-neo4j-cypher, mcp-neo4j-memory), which carry no product support or compatibility guarantees.", + "website": "https://neo4j.com/docs/mcp/current/", + "trust_vector": { + "performance_reliability": { + "overall_score": 80, + "criteria": { + "cypher_execution": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "read-cypher and write-cypher tools execute arbitrary Cypher through the official driver; read queries are validated via Neo4j's EXPLAIN-based query-type classification before execution" + } + ], + "methodology": "Cypher execution capability review", + "last_verified": "2026-07-09" + }, + "schema_introspection": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "get-schema retrieves node labels, relationship types, and property keys (APOC plugin required), with configurable schema sampling via NEO4J_SCHEMA_SAMPLE_SIZE" + } + ], + "methodology": "Schema introspection review", + "last_verified": "2026-07-09" + }, + "connection_stability": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j MCP documentation", + "url": "https://neo4j.com/docs/mcp/current/", + "date": "2026-07-09", + "value": "Connects to Aura, Neo4j Desktop, and self-managed instances via the official driver (NEO4J_URI); optional dependencies like GDS trigger adaptive mode where unsupported features disable gracefully" + } + ], + "methodology": "Connection stability and compatibility assessment", + "last_verified": "2026-07-09" + }, + "error_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "Write attempts through read-cypher are rejected by query classification and write-cypher is disabled under NEO4J_READ_ONLY=true; configurable logging (NEO4J_LOG_LEVEL, NEO4J_LOG_FORMAT) aids diagnosis" + } + ], + "methodology": "Error handling review", + "last_verified": "2026-07-09" + }, + "large_graph_handling": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "Schema sampling size is configurable for large graphs, but unbounded Cypher reads can return result sets that overwhelm model context; no automatic pagination documented" + } + ], + "methodology": "Large graph handling assessment", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 68, + "criteria": { + "authentication_security": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "neo4j-mcp-server on PyPI", + "url": "https://pypi.org/project/neo4j-mcp-server/", + "date": "2026-07-09", + "value": "Authenticates with database credentials via NEO4J_URI, NEO4J_USERNAME, NEO4J_PASSWORD, NEO4J_DATABASE environment variables; no OAuth or token-based flow for the MCP layer itself" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-07-09" + }, + "credential_exposure": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "Database username and password live in environment variables or MCP client config files on the local machine, following the same pattern as other stdio database servers" + } + ], + "methodology": "Credential security analysis", + "last_verified": "2026-07-09" + }, + "cypher_injection_risk": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "Model Context Protocol security best practices", + "url": "https://modelcontextprotocol.io/specification/2025-06-18/basic/security_best_practices", + "date": "2026-07-09", + "value": "The AI composes arbitrary Cypher; untrusted content stored in graph properties can carry injected instructions back to the model, and write-capable sessions let those instructions mutate the graph" + } + ], + "methodology": "Injection and prompt-injection exposure analysis", + "last_verified": "2026-07-09" + }, + "data_modification_control": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "neo4j-mcp-server on PyPI", + "url": "https://pypi.org/project/neo4j-mcp-server/", + "date": "2026-07-09", + "value": "write-cypher permits full create/update/delete by default; NEO4J_READ_ONLY=true disables it entirely, and read-cypher enforces read-only semantics via EXPLAIN validation — but there is no graduated gate separating writes from destructive deletes" + } + ], + "methodology": "Write-surface assessment; read-only mode is a clean kill-switch but write mode is all-or-nothing, unlike graduated designs such as ClickHouse's two-flag gate", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j query logging documentation", + "url": "https://neo4j.com/docs/operations-manual/current/monitoring/logging/", + "date": "2026-07-09", + "value": "Query logging is available in Neo4j Enterprise/Aura for agent-issued Cypher; the MCP server adds configurable structured logging but no dedicated audit trail of tool invocations" + } + ], + "methodology": "Audit logging review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 68, + "criteria": { + "graph_data_exposure": { + "score": 64, + "confidence": "high", + "evidence": [ + { + "source": "MCP Architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "Query results, schema, and sampled property values are sent to the connected LLM provider; graph schemas often encode business-sensitive relationship structures" + } + ], + "methodology": "Data flow analysis", + "last_verified": "2026-07-09" + }, + "pii_protection": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "No built-in PII detection or property-level filtering; anything readable by the configured database user can reach the model, and person-centric graphs frequently concentrate PII" + } + ], + "methodology": "PII protection assessment", + "last_verified": "2026-07-09" + }, + "self_hosted_option": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP documentation", + "url": "https://neo4j.com/docs/mcp/current/", + "date": "2026-07-09", + "value": "Runs entirely locally over stdio against self-managed Neo4j or Desktop; telemetry is controllable via NEO4J_TELEMETRY" + } + ], + "methodology": "Self-hosting options review", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Model Context Protocol documentation", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "Graph content shared with the LLM provider is governed by that provider's data terms; the server itself sends optional telemetry to Neo4j unless disabled" + } + ], + "methodology": "Data sharing analysis", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 78, + "criteria": { + "documentation_quality": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP documentation", + "url": "https://neo4j.com/docs/mcp/current/", + "date": "2026-07-09", + "value": "Dedicated official docs site covering prerequisites (APOC), configuration env vars, adaptive mode for optional dependencies, and client setup" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "Source available under the official neo4j org (268 stars, written in Go, distributed via PyPI); repo carries Apache 2.0 terms while PyPI metadata lists GPL-3.0-only — an unresolved licensing inconsistency" + } + ], + "methodology": "Source code review; scored down slightly for the conflicting license metadata between the repository and the PyPI package", + "last_verified": "2026-07-09" + }, + "official_vs_labs_clarity": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP developer guide", + "url": "https://neo4j.com/developer/genai-ecosystem/model-context-protocol-mcp/", + "date": "2026-07-09", + "value": "Neo4j documents the split explicitly: neo4j/mcp is the supported product server, while neo4j-contrib/mcp-neo4j (mcp-neo4j-cypher, mcp-neo4j-memory, aura-manager, GDS agent) are Labs projects — actively developed but experimental, with no SLAs or backwards-compatibility guarantees" + } + ], + "methodology": "Product-boundary clarity review; the split is documented but the coexisting Labs servers with overlapping names still cause user confusion", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "268 stars, 54 forks; regular releases culminating in v1.5.3 (2026-06-11), maintained by the Neo4j product team with an active Labs ecosystem alongside" + } + ], + "methodology": "Community engagement analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 78, + "criteria": { + "ease_of_setup": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "neo4j-mcp-server on PyPI", + "url": "https://pypi.org/project/neo4j-mcp-server/", + "date": "2026-07-09", + "value": "pip install neo4j-mcp-server plus four connection env vars; runs as a Python-launched subprocess over stdio against Aura, Desktop, or self-managed Neo4j (APOC required)" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "query_performance": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j performance documentation", + "url": "https://neo4j.com/docs/operations-manual/current/performance/", + "date": "2026-07-09", + "value": "Performance follows the underlying Neo4j instance: fast index-backed traversals, but unindexed agent-composed pattern matches on large graphs can be slow" + } + ], + "methodology": "Performance review of the underlying engine for agent workloads", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "Focused tool set: get-schema, read-cypher, write-cypher, list-gds-procedures; Aura management, agent memory, and GDS execution live in separate Labs servers rather than the product server" + } + ], + "methodology": "Feature coverage assessment; deliberately narrow product surface with breadth delegated to experimental Labs servers", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j MCP repository", + "url": "https://github.com/neo4j/mcp", + "date": "2026-07-09", + "value": "Product-team maintained with steady releases (v1.5.3, 2026-06-11) built on the official Go driver; adaptive mode degrades gracefully when optional dependencies are missing" + } + ], + "methodology": "Reliability analysis", + "last_verified": "2026-07-09" + }, + "community_support": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Neo4j MCP developer guide", + "url": "https://neo4j.com/developer/genai-ecosystem/model-context-protocol-mcp/", + "date": "2026-07-09", + "value": "Supported by the Neo4j product team, with the Field GenAI team maintaining the complementary Labs servers and active developer-relations content" + } + ], + "methodology": "Community support assessment", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Official product server maintained by the Neo4j product team (v1.5.3, 2026-06-11), clearly separated from experimental Labs servers", + "read-cypher enforces read-only semantics via EXPLAIN-based query classification rather than string heuristics", + "Global kill-switch: NEO4J_READ_ONLY=true fully disables write-cypher", + "First-class graph schema introspection gives agents accurate label/relationship/property context", + "Works across Aura, Neo4j Desktop, and self-managed instances with graceful adaptive mode", + "GDS procedure discovery enables graph-algorithm-aware agents" + ], + "limitations": [ + "Arbitrary Cypher means full read+write to the graph by default — write-cypher is enabled unless NEO4J_READ_ONLY=true is set", + "No graduated write gate: enabling writes also enables destructive deletes (DETACH DELETE, index/constraint drops) within the DB user's privileges", + "Database credentials stored in environment variables or client config; no OAuth flow at the MCP layer", + "Query results and graph structure exposed to the LLM provider with no PII filtering", + "License metadata inconsistent: repo indicates Apache 2.0 while PyPI declares GPL-3.0-only", + "Requires the APOC plugin; overlapping Labs servers (mcp-neo4j-cypher/memory) with similar names invite confusion", + "stdio-only product server; no managed remote endpoint from Neo4j yet" + ], + "metadata": { + "license": "Apache 2.0 per repository LICENSE; PyPI metadata lists GPL-3.0-only (inconsistency noted)", + "supported_platforms": ["All platforms with Python 3.10+ (Go binary distributed via PyPI)"], + "programming_languages": ["Go"], + "mcp_version": "1.0", + "github_repo": "https://github.com/neo4j/mcp", + "github_stars": 268, + "package_name": "neo4j-mcp-server (PyPI)", + "latest_version": "1.5.3 (2026-06-11)", + "docs": "https://neo4j.com/docs/mcp/current/", + "api_dependency": "Neo4j Go driver; APOC plugin required; optional Graph Data Science library", + "authentication": "Database credentials via NEO4J_URI / NEO4J_USERNAME / NEO4J_PASSWORD / NEO4J_DATABASE env vars", + "security_controls": ["NEO4J_READ_ONLY=true disables write-cypher", "EXPLAIN-based read-only validation for read-cypher", "configurable telemetry (NEO4J_TELEMETRY)"], + "labs_counterparts": "neo4j-contrib/mcp-neo4j: mcp-neo4j-cypher, mcp-neo4j-memory, mcp-neo4j-aura-manager, GDS agent (experimental, no SLAs)", + "first_release": "2025 (Labs servers); product server GA line through 2026, v1.5.3 on 2026-06-11", + "transport_types": ["stdio"], + "maintained_by": "Neo4j product team" + }, + "use_case_ratings": { + "code-generation": { + "overall": 78, + "notes": "Good for generating Cypher and graph data models from live schemas" + }, + "customer-support": { + "overall": 68, + "notes": "Useful over customer-360 graphs, but such graphs concentrate PII" + }, + "content-creation": { + "overall": 62, + "notes": "Knowledge-graph-backed content research is a reasonable secondary fit" + }, + "data-analysis": { + "overall": 86, + "notes": "Strong for relationship-centric analysis, fraud rings, and network exploration" + }, + "research-assistant": { + "overall": 84, + "notes": "Knowledge graphs plus GDS discovery suit multi-hop research questions well" + }, + "legal-compliance": { + "overall": 58, + "notes": "Entity-relationship discovery helps investigations; PII exposure needs strict scoping" + }, + "healthcare": { + "overall": 50, + "notes": "Patient/provider graphs are PII-dense; not advised without read-only mode and scoped users" + }, + "financial-analysis": { + "overall": 74, + "notes": "Well suited to fraud and exposure network analysis with read-only credentials" + }, + "education": { + "overall": 84, + "notes": "Excellent for teaching graph modeling and Cypher interactively" + }, + "creative-writing": { + "overall": 66, + "notes": "Character/world relationship graphs are a genuinely good niche fit" + } + }, + "best_for": [ + "Teams building agents over knowledge graphs, fraud networks, and recommendation graphs", + "Developers wanting official, product-supported Neo4j access instead of experimental Labs servers", + "Read-heavy graph exploration with NEO4J_READ_ONLY=true and a least-privilege database user", + "GraphRAG and multi-hop reasoning workloads grounded in explicit schema" + ], + "related_entities": ["mcp-server-memory", "mcp-server-postgres", "mcp-server-mongodb", "mcp-server-clickhouse"], + "tags": ["database", "graph", "cypher", "knowledge-graph", "mcp", "model-context-protocol", "neo4j"] +} diff --git a/data/mcps/mcp-server-neon.json b/data/mcps/mcp-server-neon.json new file mode 100644 index 0000000..5a75085 --- /dev/null +++ b/data/mcps/mcp-server-neon.json @@ -0,0 +1,445 @@ +{ + "id": "mcp-server-neon", + "type": "mcp", + "name": "MCP Neon Server", + "provider": "Neon (Databricks)", + "version": "hosted remote (mcp.neon.tech); npm 0.6.5 deprecated", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Official Neon MCP server for managing serverless Postgres via natural language: projects, branches, SQL execution, and branch-based safe migrations (20+ tools). Remote-only since Feb 2026 at https://mcp.neon.tech/mcp with OAuth (or API-key header); the local stdio CLI was removed and npm @neondatabase/mcp-server-neon (0.6.5) is deprecated. Full DDL/DML capability means Neon (Databricks-owned since May 2025) recommends development/testing use, not production databases.", + "website": "https://neon.com/docs/ai/neon-mcp-server", + "trust_vector": { + "performance_reliability": { + "overall_score": 87, + "criteria": { + "api_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Neon MCP Server docs", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "Hosted server fronts the Neon Management API and Postgres; managed by Neon with automatic updates, removing local-version drift" + } + ], + "methodology": "API stability review of the hosted endpoint and underlying Neon platform", + "last_verified": "2026-07-09" + }, + "sql_execution": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Neon MCP Server docs - tools", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "run_sql and transaction tools execute full Postgres queries and DDL/DML against any project branch, with schema-inspection tools for tables and columns" + } + ], + "methodology": "SQL execution capability review", + "last_verified": "2026-07-09" + }, + "branching_operations": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Databricks-Neon acquisition announcement", + "url": "https://www.databricks.com/blog/databricks-neon", + "date": "2025-05-14", + "value": "Neon's copy-on-write architecture spins up Postgres branches in under half a second; MCP tools create branches, compare schemas, and reset to parent state" + } + ], + "methodology": "Branch tool review against Neon's instant-branching architecture", + "last_verified": "2026-07-09" + }, + "migration_safety": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Neon MCP Server docs - migrations", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "Schema changes use a safe temporary-branch workflow: migrations are prepared and tested on a throwaway branch before being applied to the target, reducing the chance of destructive mistakes" + } + ], + "methodology": "Review of the prepare/complete migration workflow design", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 77, + "criteria": { + "authentication_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Neon blog - Bringing MCP to the Cloud", + "url": "https://neon.com/blog/bringing-mcp-to-the-cloud", + "date": "2025-04-11", + "value": "Remote server acts as an OAuth client to Neon and OAuth server to MCP clients: users authorize via their Neon account instead of placing long-lived API keys in plaintext config files; API-key Authorization header remains available for non-interactive use" + } + ], + "methodology": "Authentication mechanism review of the OAuth flow and API-key fallback", + "last_verified": "2026-07-09" + }, + "prompt_injection_sql_risk": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Neon MCP Server docs - security guidance", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "Docs instruct: 'Always review and authorize LLM-requested actions before execution. Restrict MCP access to trusted users and regularly audit access.' An agent reading untrusted rows while holding full SQL access remains exposed to the prompt-injection/data-leak class demonstrated against database MCPs in 2025; no result-wrapping mitigation is documented" + } + ], + "methodology": "Prompt-injection and SQL execution risk assessment relative to database MCP attack research", + "last_verified": "2026-07-09" + }, + "destructive_operation_control": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Neon MCP Server docs - tools", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "Tools include project deletion, branch deletion, and unrestricted DDL/DML; branch-based migration flow adds a safety layer for schema changes, but destructive actions are otherwise gated only by client-side tool approval and URL-parameter access controls" + } + ], + "methodology": "Capability analysis of destructive tools and available guardrails", + "last_verified": "2026-07-09" + }, + "key_management": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Neon blog - Bringing MCP to the Cloud", + "url": "https://neon.com/blog/bringing-mcp-to-the-cloud", + "date": "2025-04-11", + "value": "Motivation for the remote model explicitly cited eliminating plaintext API keys in configuration files; OAuth grants are revocable from the Neon account, and hosted operation removes stale-version credential handling" + } + ], + "methodology": "Key management review of OAuth-first credential model", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 68, + "criteria": { + "data_exposure_to_llm": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "MCP data flow architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "SQL result sets, schema definitions, and connection metadata returned by tools are sent to the LLM provider as tool results" + } + ], + "methodology": "Data flow analysis of query and schema tool outputs", + "last_verified": "2026-07-09" + }, + "data_residency": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Neon MCP Server docs", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "All MCP traffic now transits Neon's hosted server at mcp.neon.tech regardless of client; database regions are selectable in Neon, but there is no self-hosted MCP path since the stdio removal" + } + ], + "methodology": "Data residency review of the remote-only architecture", + "last_verified": "2026-07-09" + }, + "platform_compliance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Neon privacy and compliance", + "url": "https://neon.com/privacy-policy", + "date": "2026-07-09", + "value": "Neon platform (now part of Databricks) maintains SOC 2 and GDPR-aligned data processing commitments; MCP access inherits the platform's compliance posture" + } + ], + "methodology": "Platform compliance documentation review", + "last_verified": "2026-07-09" + }, + "self_hosted_option": { + "score": 55, + "confidence": "high", + "evidence": [ + { + "source": "neondatabase/mcp-server-neon issue #214", + "url": "https://github.com/neondatabase/mcp-server-neon/issues/214", + "date": "2026-07-09", + "value": "Local stdio CLI removed (Feb 2026, PR #198) and npm package @neondatabase/mcp-server-neon (0.6.5) deprecated; users report headless/gateway deployments broken with no self-hosted migration path" + } + ], + "methodology": "Self-hosting options review after the remote-only transition", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 82, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Neon docs - AI section", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "Clear official documentation covering setup, OAuth and API-key auth, full tool reference by category, safety guidance, and explicit development-only recommendation" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "neondatabase/mcp-server-neon repository", + "url": "https://github.com/neondatabase/mcp-server-neon", + "date": "2026-07-09", + "value": "Server source is MIT-licensed and public (617 stars, 115 forks), so the remote server's tool logic remains inspectable even though only Neon operates it" + } + ], + "methodology": "Source code and license review", + "last_verified": "2026-07-09" + }, + "deprecation_communication": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "neondatabase/mcp-server-neon issue #214", + "url": "https://github.com/neondatabase/mcp-server-neon/issues/214", + "date": "2026-07-09", + "value": "The forced local-to-remote shift deprecated the npm package and removed stdio with limited notice; CI/CD and gateway users were left without a supported migration path, an open trust concern for teams that depended on the local server" + } + ], + "methodology": "Change management and deprecation communication assessment", + "last_verified": "2026-07-09" + }, + "vendor_credibility": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Databricks press release", + "url": "https://www.databricks.com/company/newsroom/press-releases/databricks-agrees-acquire-neon-help-developers-deliver-ai-systems", + "date": "2025-05-14", + "value": "Neon acquired by Databricks for ~$1B (announced 2025-05-14); the platform is officially maintained with strong backing and an AI-agent-first roadmap (80%+ of Neon databases were created by agents)" + } + ], + "methodology": "Maintainer reputation and corporate backing analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 81, + "criteria": { + "ease_of_setup": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Neon MCP Server docs - setup", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "Add https://mcp.neon.tech/mcp and complete the browser OAuth flow — no installation, no API key handling, automatic updates" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "operation_coverage": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Neon MCP Server docs - tool reference", + "url": "https://neon.com/docs/ai/neon-mcp-server", + "date": "2026-07-09", + "value": "20+ tools across projects/organizations, branches (create, compare schemas, reset), SQL execution and transactions, safe migrations, query optimization, Neon Auth provisioning, Data API setup, and documentation search" + } + ], + "methodology": "Feature coverage assessment against database management workflows", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Neon status and platform", + "url": "https://neonstatus.com/", + "date": "2026-07-09", + "value": "Hosted MCP rides Neon's managed platform; single hosted endpoint is a shared dependency but removes local-version breakage" + } + ], + "methodology": "Reliability analysis of the hosted deployment model", + "last_verified": "2026-07-09" + }, + "ci_cd_compatibility": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "neondatabase/mcp-server-neon issue #214", + "url": "https://github.com/neondatabase/mcp-server-neon/issues/214", + "date": "2026-07-09", + "value": "OAuth-first remote server complicates headless, gateway, and CI/CD deployments; users report the stdio deprecation broke automated pipelines, with API-key header auth the remaining workaround" + } + ], + "methodology": "Non-interactive deployment compatibility assessment", + "last_verified": "2026-07-09" + }, + "maintenance_activity": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "neondatabase/mcp-server-neon repository", + "url": "https://github.com/neondatabase/mcp-server-neon", + "date": "2026-07-09", + "value": "Active development on the remote server with hosted auto-updates; local npm package frozen at deprecated 0.6.5" + } + ], + "methodology": "Repository activity and release model review", + "last_verified": "2026-07-09" + } + } + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 90, + "notes": "Excellent for scaffolding schemas, migrations, and app databases during development" + }, + "customer-support": { + "overall": 70, + "notes": "Can inspect data for support workflows, but production access is discouraged" + }, + "content-creation": { + "overall": 72, + "notes": "Useful for spinning up content backends and prototypes quickly" + }, + "data-analysis": { + "overall": 85, + "notes": "Full SQL over Postgres branches; branch snapshots enable safe analysis copies" + }, + "research-assistant": { + "overall": 78, + "notes": "Good for building research datasets on disposable branches" + }, + "legal-compliance": { + "overall": 62, + "notes": "Remote-only MCP path and LLM data exposure complicate strict compliance regimes" + }, + "healthcare": { + "overall": 58, + "notes": "Not recommended: PHI would transit Neon's MCP endpoint and the LLM provider" + }, + "financial-analysis": { + "overall": 70, + "notes": "Usable on non-production branches; avoid live financial data" + }, + "education": { + "overall": 88, + "notes": "Great for learning Postgres — instant free branches make experimentation safe" + }, + "creative-writing": { + "overall": 55, + "notes": "Limited relevance beyond storing creative project data" + } + }, + "best_for": [ + "Developers building and iterating on Postgres schemas with AI assistance", + "Agentic workflows that provision databases and branches on demand", + "Teams using branch-based development to keep AI changes off production", + "Rapid prototyping where OAuth setup and zero install matter" + ], + "not_recommended_for": [ + "Production database administration (explicitly discouraged by Neon)", + "Headless CI/CD or MCP-gateway deployments that need a local stdio server", + "Workloads with regulated data that cannot transit Neon's hosted MCP endpoint" + ], + "strengths": [ + "OAuth-based remote server: no API keys in config files, revocable grants, automatic updates", + "Branch-based safe migration workflow tests schema changes on throwaway branches first", + "Sub-second copy-on-write branching makes disposable AI sandboxes practical", + "20+ tools covering projects, branches, SQL, migrations, query optimization, and Neon Auth", + "Backed by Databricks (acquired ~$1B, May 2025) with an AI-agent-first roadmap", + "MIT-licensed source remains public and inspectable" + ], + "limitations": [ + "Remote-only since Feb 2026: stdio removed, npm package deprecated, no self-hosted option", + "Forced local-to-remote migration broke headless/gateway/CI-CD deployments (issue #214)", + "Full DDL/DML on any connected database — destructive actions gated mainly by client approval", + "No documented prompt-injection mitigations (e.g., result wrapping or read-only mode) unlike some database MCP peers", + "Query results and schemas exposed to the LLM provider", + "Neon explicitly recommends development/testing use only, not production", + "All MCP traffic transits Neon's hosted endpoint — a shared availability and trust dependency" + ], + "metadata": { + "license": "MIT (source); hosted service operated by Neon", + "supported_platforms": [ + "Any MCP client supporting streamable HTTP (remote)" + ], + "programming_languages": [ + "TypeScript" + ], + "mcp_version": "1.0", + "github_repo": "https://github.com/neondatabase/mcp-server-neon", + "github_stars": 617, + "remote_endpoint": "https://mcp.neon.tech/mcp", + "deprecated_package": "@neondatabase/mcp-server-neon (0.6.5, deprecated; stdio CLI removed Feb 2026)", + "api_dependency": "Neon Management API / Postgres", + "authentication": "OAuth (browser flow, default); Neon API key via Authorization header for non-interactive use", + "first_release": "2024-12", + "remote_launch": "2025-04-11", + "remote_only_since": "2026-02", + "maintained_by": "Neon (Databricks)", + "status": "Active - remote-only hosted server", + "transport_types": [ + "streamable-http (hosted)" + ], + "installation_methods": [ + "Remote MCP endpoint (OAuth)", + "Remote MCP endpoint (API-key header)" + ] + }, + "related_entities": [ + "mcp-server-supabase", + "mcp-server-postgres", + "mcp-server-mongodb" + ], + "tags": [ + "database", + "postgresql", + "serverless", + "neon", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ] +} diff --git a/data/mcps/mcp-server-paypal.json b/data/mcps/mcp-server-paypal.json new file mode 100644 index 0000000..2f298c6 --- /dev/null +++ b/data/mcps/mcp-server-paypal.json @@ -0,0 +1,447 @@ +{ + "id": "mcp-server-paypal", + "type": "mcp", + "name": "PayPal MCP Server", + "provider": "PayPal", + "version": "2026.7", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "PayPal's official MCP server (Agent Toolkit), hosted at https://mcp.paypal.com with SSE (/sse) and streamable HTTP (/http) transports plus a sandbox at mcp.sandbox.paypal.com; also installable locally as @paypal/mcp. OAuth login or client-credential access tokens gate access, with restricted tool visibility so the LLM only sees tools its token permits. Tools cover invoices, orders, captures, refunds, disputes, subscriptions, catalog, shipment tracking, and transaction reporting.", + "website": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "trust_vector": { + "performance_reliability": { + "overall_score": 84, + "criteria": { + "api_reliability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "PayPal Status", + "url": "https://www.paypal-status.com/product/production", + "date": "2026-07-09", + "value": "Built directly on PayPal's production REST APIs, which serve global payment volume with historically high availability and a public status page" + } + ], + "methodology": "API stability and uptime analysis of the underlying PayPal REST APIs", + "last_verified": "2026-07-09" + }, + "operation_success_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal MCP Server Repository", + "url": "https://github.com/paypal/paypal-mcp-server", + "date": "2026-07-09", + "value": "Tools are wrappers over stable PayPal API endpoints across invoicing (create, list, send, remind, cancel, QR codes), orders/captures/refunds, disputes, subscriptions, catalog, tracking, and transaction reporting" + } + ], + "methodology": "Operation success testing across the documented tool set in sandbox mode", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal REST API Rate Limiting Guidance", + "url": "https://developer.paypal.com/reference/guidelines/rate-limiting/", + "date": "2026-07-09", + "value": "Inherits PayPal REST API rate limiting with standard 429 responses; limits are adequate for typical agent workloads though not published as fixed quotas" + } + ], + "methodology": "Rate limiting behavior testing under sustained tool-call load", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal MCP Server Repository", + "url": "https://github.com/paypal/paypal-mcp-server", + "date": "2026-07-09", + "value": "Surfaces PayPal's structured error objects (validation errors, authorization failures) to the agent; strict schema validation rejects malformed tool requests before they reach the API" + } + ], + "methodology": "Error handling testing with invalid parameters, insufficient permissions, and declined operations", + "last_verified": "2026-07-09" + }, + "sandbox_environment_fidelity": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "A full parallel sandbox server at https://mcp.sandbox.paypal.com mirrors the production tool surface, enabling safe end-to-end agent evaluation before any live money movement" + } + ], + "methodology": "Parity comparison of sandbox and production tool surfaces and behaviors", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 78, + "criteria": { + "authentication_security": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "Remote server supports OAuth (user logs into PayPal and grants consent) or access tokens minted from Developer Dashboard client credentials; sandbox and production are fully separated hosts with separate credentials" + } + ], + "methodology": "Authentication mechanism review for OAuth and access-token flows on the hosted endpoints", + "last_verified": "2026-07-09" + }, + "scope_limitation": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "PayPal Community Blog: Expanding PayPal's Remote MCP Server Tools", + "url": "https://developer.paypal.com/community/blog/mcp-server-toolexpansion/", + "date": "2026-07-09", + "value": "Restricted tool visibility ensures only authorized users and models can access specific tools: the tool list itself is filtered server-side by token permissions, so the LLM never even sees tools its token does not permit — a stronger design than post-hoc authorization checks" + } + ], + "methodology": "Permission scope testing across tokens with differing grants; verification that unpermitted tools are absent from tools/list responses", + "last_verified": "2026-07-09" + }, + "token_exposure_risk": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "Local @paypal/mcp mode requires client ID and secret in client configuration; a leaked live credential grants API access at the credential's full permission level, making OAuth on the remote server the safer default" + } + ], + "methodology": "Token storage and exposure-surface analysis for local configuration versus hosted OAuth", + "last_verified": "2026-07-09" + }, + "action_auditability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal Developer Dashboard", + "url": "https://developer.paypal.com/dashboard/", + "date": "2026-07-09", + "value": "Tool calls are API requests attributable to the app credential or OAuth grant, visible in Developer Dashboard event logs and account transaction/activity history, though less granular than a dedicated per-tool-call audit log" + } + ], + "methodology": "Audit logging review of API event logs and account activity attribution", + "last_verified": "2026-07-09" + }, + "unauthorized_action_risk": { + "score": 58, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "Tool surface includes direct money movement (refunds, order capture, invoicing, subscription changes) with no server-side human confirmation step; restricted tool visibility narrows the blast radius but a prompt-injected agent holding a write-permitted token can still move real funds" + } + ], + "methodology": "Threat modeling of write-capable financial tools under prompt injection; calibrated against other payment-domain MCP servers", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 75, + "criteria": { + "payment_data_exposure": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Server Repository", + "url": "https://github.com/paypal/paypal-mcp-server", + "date": "2026-07-09", + "value": "Invoice recipients, order payer details, dispute records, and transaction reports returned by tools flow into the LLM provider's context; raw card credentials are never exposed by PayPal's REST APIs" + } + ], + "methodology": "Data flow analysis of tool results containing payer and transaction data", + "last_verified": "2026-07-09" + }, + "sensitive_data_protection": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "PayPal Security Center", + "url": "https://www.paypal.com/us/security", + "date": "2026-07-09", + "value": "PayPal operates PCI-DSS compliant payment infrastructure; card and bank credentials are tokenized server-side and are not retrievable through any MCP tool" + } + ], + "methodology": "Review of PayPal compliance posture and the data classes reachable via the tool surface", + "last_verified": "2026-07-09" + }, + "organization_data_control": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "Account owners control exposure through app credential permissions (which drive restricted tool visibility), sandbox/production separation, and revocation of OAuth consents and app credentials in the Developer Dashboard" + } + ], + "methodology": "Access control review of credential permissioning and consent revocation", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal Privacy Statement", + "url": "https://www.paypal.com/us/legalhub/privacy-full", + "date": "2026-07-09", + "value": "Payer PII retrieved via tools is shared with the user's LLM provider under that provider's data policy, outside PayPal's compliance boundary" + } + ], + "methodology": "Analysis of downstream data sharing once tool results leave PayPal", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 80, + "criteria": { + "documentation_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "Dedicated quickstart covering remote (SSE and streamable HTTP) and local setup, both auth flows, and client configuration for Claude, Cursor, and Cline; tool catalog lives in a separate agent-tools reference, and material is split across docs.paypal.ai and developer.paypal.com" + } + ], + "methodology": "Documentation completeness and accuracy review across the developer portals", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal Developer Dashboard", + "url": "https://developer.paypal.com/dashboard/", + "date": "2026-07-09", + "value": "Operations surface in Developer Dashboard event logs and account activity; financial writes (invoices, refunds, captures) additionally appear in the account's transaction history" + } + ], + "methodology": "Logging and traceability assessment via dashboard logs and transaction records", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Server Repository", + "url": "https://github.com/paypal/paypal-mcp-server", + "date": "2026-07-09", + "value": "Apache-2.0 licensed server source (@paypal/mcp v1.8.1 on npm) is published and auditable, though community traction is thin (11 stars) and the hosted remote server's exact deployment cannot be independently verified against the repo" + } + ], + "methodology": "Source code review of the published package and comparison against observed hosted behavior", + "last_verified": "2026-07-09" + }, + "api_coverage_clarity": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal Community Blog: Expanding PayPal's Remote MCP Server Tools", + "url": "https://developer.paypal.com/community/blog/mcp-server-toolexpansion/", + "date": "2026-07-09", + "value": "Remote server expanded from invoicing-only at launch (April 2025) to the full agent toolkit (June 2025): invoices, orders, captures, refunds, disputes, subscriptions and plans, catalog, shipment tracking, and transaction reporting" + } + ], + "methodology": "Comparison of documented tool surface against the shipped package and hosted tools/list output", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 80, + "criteria": { + "ease_of_setup": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "Remote endpoints connect via OAuth with a simple PayPal login (some clients require the mcp-remote shim); local mode is an npm install of @paypal/mcp with Node.js 18+ and dashboard credentials" + } + ], + "methodology": "Setup complexity assessment for hosted and local transports across major MCP clients", + "last_verified": "2026-07-09" + }, + "api_performance": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal MCP Quickstart", + "url": "https://docs.paypal.ai/developer/tools/ai/mcp-quickstart", + "date": "2026-07-09", + "value": "Streamable HTTP transport (/http) delivers chunked responses for immediate display; underlying REST API latency is typical of production payment APIs with thin wrapper overhead" + } + ], + "methodology": "Latency observation across representative tool calls on both transports", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal Status", + "url": "https://www.paypal-status.com/product/production", + "date": "2026-07-09", + "value": "Hosted MCP and underlying APIs ride PayPal's production payments infrastructure with strong historical availability" + } + ], + "methodology": "Uptime analysis of PayPal infrastructure hosting the MCP endpoints", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "PayPal MCP Server Repository", + "url": "https://github.com/paypal/paypal-mcp-server", + "date": "2026-07-09", + "value": "Covers the core merchant surface (invoicing, orders, refunds, disputes, subscriptions, catalog, tracking, reporting); payouts, Braintree, and marketplace/platform features are not exposed" + } + ], + "methodology": "Feature completeness assessment against the full PayPal API surface", + "last_verified": "2026-07-09" + }, + "community_adoption": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "PayPal MCP Server Repository", + "url": "https://github.com/paypal/paypal-mcp-server", + "date": "2026-07-09", + "value": "First-party maintained, but community traction is modest: 11 GitHub stars, last repository push October 2025, and @paypal/mcp last published to npm October 2025; most usage flows through the hosted remote server" + } + ], + "methodology": "Community activity and adoption analysis across GitHub, npm, and MCP client ecosystems", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Restricted tool visibility: the LLM only sees tools its token permits, filtered server-side — a best-in-class scoping design", + "Hosted remote server with OAuth avoids placing client secrets in local config", + "Full parallel sandbox environment (mcp.sandbox.paypal.com) for safe end-to-end agent testing", + "Broad merchant tool surface: invoices, orders, refunds, disputes, subscriptions, catalog, tracking, reporting", + "Apache-2.0 open-source server implementation published as @paypal/mcp", + "Both SSE and streamable HTTP transports on the hosted endpoint", + "Backed by PCI-DSS compliant production payments infrastructure" + ], + "limitations": [ + "Money-movement tools (refunds, captures, invoicing, subscription changes) make misuse directly costly", + "No server-side human confirmation step on write operations; must be enforced by the client", + "Payer PII in tool results is shared with the LLM provider", + "Client ID/secret in local stdio config is a high-value leak target", + "No dedicated per-tool-call audit log beyond standard API event logs", + "Local package and repository updates have lagged the hosted server (last npm publish October 2025)", + "Payouts, Braintree, and marketplace features are not exposed" + ], + "metadata": { + "repository": "https://github.com/paypal/paypal-mcp-server", + "package_name": "@paypal/mcp", + "package_version": "1.8.1", + "license": "Apache-2.0", + "maintained_by": "PayPal", + "github_stars": 11, + "remote_endpoint": "https://mcp.paypal.com/sse (SSE); https://mcp.paypal.com/http (streamable HTTP); sandbox at https://mcp.sandbox.paypal.com", + "authentication": "OAuth (hosted, recommended) or access tokens from Developer Dashboard client credentials; restricted tool visibility filters the tool list by token permissions", + "transport_types": [ + "sse (hosted)", + "streamable-http (hosted)", + "stdio (@paypal/mcp, local)" + ], + "installation_methods": [ + "Remote MCP endpoint (OAuth or bearer token, mcp-remote shim where needed)", + "npm @paypal/mcp (Node.js 18+)" + ], + "compliance": [ + "PCI-DSS" + ], + "mcp_version": "1.0" + }, + "use_case_ratings": { + "financial-analysis": { + "overall": 78, + "notes": "Transaction reporting and dispute/subscription queries directly from the source of truth; weaker than dedicated analytics for bulk work" + }, + "customer-support": { + "overall": 84, + "notes": "Strong for support agents managing invoices, order lookups, refunds, and dispute status with permission-scoped tokens" + }, + "code-generation": { + "overall": 74, + "notes": "Sandbox parity makes integration development safe, though there is no built-in docs-search tool" + }, + "data-analysis": { + "overall": 70, + "notes": "Good for ad-hoc transaction and subscription queries; bulk analytics is better served by reporting exports" + }, + "legal-compliance": { + "overall": 62, + "notes": "Dispute records are accessible, but payer PII flowing to LLM providers requires careful review" + } + }, + "best_for": [ + "Merchant support and operations teams automating invoicing, refund, and dispute workflows with permission-scoped tokens", + "Developers building and testing PayPal integrations against the sandbox MCP before going live", + "Finance teams querying transactions, subscriptions, and disputes conversationally", + "Agent builders who want server-side tool scoping rather than client-side-only guardrails" + ], + "not_recommended_for": [ + "Fully autonomous agents holding write-permitted live tokens without human confirmation", + "Payout, marketplace, or Braintree workflows (not exposed via MCP)", + "Environments requiring granular per-tool-call audit trails" + ], + "related_entities": [ + "mcp-server-stripe", + "mcp-server-zapier", + "mcp-server-slack", + "mcp-server-github" + ], + "tags": [ + "payments", + "invoicing", + "fintech", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-salesforce.json b/data/mcps/mcp-server-salesforce.json new file mode 100644 index 0000000..9bbb9e4 --- /dev/null +++ b/data/mcps/mcp-server-salesforce.json @@ -0,0 +1,492 @@ +{ + "id": "mcp-server-salesforce", + "type": "mcp", + "name": "MCP Salesforce Server", + "provider": "Salesforce (Official)", + "version": "hosted (GA 2026-04-29); DX @salesforce/mcp 0.30.14", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Salesforce's official MCP offering. Primary path is Hosted MCP Servers (GA 2026-04-29, included for Enterprise Edition+ and Developer Edition orgs): Salesforce-managed endpoints exposing org data, flows, Apex actions, and Named Query APIs via OAuth with PKCE, enforcing CRUD/FLS/sharing as the authenticated user. Variants: local DX MCP server (@salesforce/mcp, v0.30.x, Apache-2.0) for dev workflows, and the Data 360 MCP server in developer preview (May 2026).", + "website": "https://developer.salesforce.com/docs/platform/hosted-mcp-servers/guide/servers-reference.html", + "trust_vector": { + "performance_reliability": { + "overall_score": 84, + "criteria": { + "crm_operation_reliability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Salesforce hosts and scales the MCP servers on the same managed infrastructure as its REST APIs; structured tool calls replace unstructured API access" + } + ], + "methodology": "Assessment of hosted-server operation reliability against the underlying Salesforce API platform", + "last_verified": "2026-07-09" + }, + "query_and_flow_execution": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Salesforce Hosted MCP Servers - Standard Servers Reference", + "url": "https://developer.salesforce.com/docs/platform/hosted-mcp-servers/guide/servers-reference.html", + "date": "2026-07-09", + "value": "Prebuilt standard servers (Agentforce 360 Platform, Tableau Next, Data 360 SQL) plus custom servers exposing flows, Apex actions, and Named Query APIs as defined tool sets" + } + ], + "methodology": "Review of tool execution paths for queries, flows, and Apex actions", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Salesforce API Request Limits", + "url": "https://developer.salesforce.com/docs/atlas.en-us.salesforce_app_limits_cheatsheet.meta/salesforce_app_limits_cheatsheet/salesforce_app_limits_platform_api.htm", + "date": "2026-07-09", + "value": "MCP operations consume org API capacity subject to per-org API request limits by edition" + } + ], + "methodology": "Analysis of org API limit consumption by MCP tool calls", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Fully managed infrastructure: no servers to provision, no uptime to manage; Salesforce scales the servers as it does its REST APIs" + } + ], + "methodology": "Managed-infrastructure scalability assessment", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Salesforce Hosted MCP Servers Help", + "url": "https://help.salesforce.com/s/articleView?id=platform.hosted_mcp_servers.htm&language=en_US&type=5", + "date": "2026-07-09", + "value": "Tool calls surface standard Salesforce API errors (permission, validation, limits) to the client for recovery" + } + ], + "methodology": "Error propagation and recovery behavior review", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 81, + "criteria": { + "authentication_security": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "OAuth with PKCE controls access; a new MCP-specific OAuth scope separates MCP access from existing REST API scopes, granted via an External Client App" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-07-09" + }, + "permission_scope_control": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Existing org permissions automatically apply - CRUD, field-level security, and sharing rules; every transaction runs as the authenticated user with no anonymous service accounts" + } + ], + "methodology": "Access control model analysis", + "last_verified": "2026-07-09" + }, + "data_modification_risk": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Hosted MCP Servers - Standard Servers Reference", + "url": "https://developer.salesforce.com/docs/platform/hosted-mcp-servers/guide/servers-reference.html", + "date": "2026-07-09", + "value": "Standard and custom servers can execute writes (record CRUD, flows, Apex actions) with the user's full permissions; a read-only platform/sobject-reads server is available for lower-risk use such as sandboxes" + } + ], + "methodology": "Write-path risk assessment across standard and custom servers", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Full audit trails are included; MCP transactions are attributable to the authenticated user in org auditing" + } + ], + "methodology": "Audit logging review", + "last_verified": "2026-07-09" + }, + "secure_default_posture": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Secure-by-default: servers are off until an admin explicitly enables them in Setup > API Catalog > MCP Servers and creates an External Client App with appropriate scopes" + } + ], + "methodology": "Default configuration and enablement review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 68, + "criteria": { + "crm_pii_exposure": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "MCP Data Flow", + "url": "https://modelcontextprotocol.io/docs/architecture", + "date": "2026-07-09", + "value": "Contact, lead, account, and opportunity records - CRM PII at enterprise scale - are returned to the MCP client and sent to the LLM provider" + } + ], + "methodology": "Data flow analysis of CRM record exposure to the model context", + "last_verified": "2026-07-09" + }, + "least_privilege_support": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Exposure can be narrowed with permission sets, FLS, sharing rules, and by choosing which servers/tools to enable (including a read-only server)" + } + ], + "methodology": "Assessment of mechanisms available to minimize data exposed to agents", + "last_verified": "2026-07-09" + }, + "data_residency_and_control": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Salesforce Hosted MCP Servers Help", + "url": "https://help.salesforce.com/s/articleView?id=platform.hosted_mcp_servers.htm&language=en_US&type=5", + "date": "2026-07-09", + "value": "MCP endpoints run inside Salesforce-managed infrastructure under existing org trust and compliance boundaries, but responses leave that boundary once delivered to the MCP client" + } + ], + "methodology": "Data residency and control boundary analysis", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 66, + "confidence": "high", + "evidence": [ + { + "source": "LLM Provider Policies", + "url": "https://modelcontextprotocol.io/docs", + "date": "2026-07-09", + "value": "CRM data retrieved via MCP is processed by the connected LLM provider (Claude, ChatGPT, Slack AI, etc.) under that provider's privacy policy" + } + ], + "methodology": "Data sharing analysis", + "last_verified": "2026-07-09" + }, + "consent_and_regulatory": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Customer PII in CRM records may be subject to GDPR/CCPA; routing it through agent contexts requires customer-side controls beyond what the server enforces" + } + ], + "methodology": "Regulatory exposure assessment for CRM data in agent workflows", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 82, + "criteria": { + "documentation_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Hosted MCP Servers Documentation", + "url": "https://developer.salesforce.com/docs/platform/hosted-mcp-servers/guide/servers-reference.html", + "date": "2026-07-09", + "value": "First-party developer guide with a per-server tool reference, setup guides (Setup > API Catalog > MCP Servers), and client connection walkthroughs including Claude" + }, + { + "source": "Salesforce Developers Blog - Connect Claude with Hosted MCP Servers", + "url": "https://developer.salesforce.com/blogs/2026/05/connect-claude-with-salesforce-hosted-mcp-servers", + "date": "2026-05-15", + "value": "Dedicated client integration guides published post-GA" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Servers define exactly which operations are available as structured tool calls, with full audit trails attributable to the authenticated user" + } + ], + "methodology": "Logging and traceability assessment", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 74, + "confidence": "high", + "evidence": [ + { + "source": "npm - @salesforce/mcp", + "url": "https://www.npmjs.com/package/@salesforce/mcp", + "date": "2026-06-23", + "value": "Hosted servers are closed-source managed services, but the DX MCP server (@salesforce/mcp v0.30.14, Apache-2.0, last published 2026-06-23) and the Data 360 MCP server (forcedotcom/d360-mcp-server) are open source" + }, + { + "source": "GitHub - forcedotcom/d360-mcp-server", + "url": "https://github.com/forcedotcom/d360-mcp-server", + "date": "2026-07-09", + "value": "Data 360 MCP server published as an open-source developer preview" + } + ], + "methodology": "Source availability review across the hosted, DX, and Data 360 variants", + "last_verified": "2026-07-09" + }, + "api_coverage_clarity": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Data 360 MCP Server (Developer Preview)", + "url": "https://developer.salesforce.com/blogs/2026/05/introducing-the-data-360-mcp-server-developer-preview", + "date": "2026-05-20", + "value": "Coverage is explicitly documented per server; the Data 360 preview consolidates ~200 REST operations behind three facade tools (search, payload_examples, execute) to avoid context-window overload" + } + ], + "methodology": "API coverage documentation review", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 81, + "criteria": { + "ease_of_setup": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Developers Blog - Hosted MCP Servers GA", + "url": "https://developer.salesforce.com/blogs/2026/04/salesforce-hosted-mcp-servers-are-now-generally-available", + "date": "2026-04-29", + "value": "Setup takes under 30 minutes but requires admin work: enable a server in Setup > API Catalog > MCP Servers, create an External Client App with MCP scopes, then connect the client via OAuth" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "api_performance": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Salesforce Developer Platform", + "url": "https://developer.salesforce.com/", + "date": "2026-07-09", + "value": "Performance tracks the underlying Salesforce APIs; complex queries and flow executions add latency versus simple record reads" + } + ], + "methodology": "Performance assessment against underlying API behavior", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Trust Status", + "url": "https://status.salesforce.com/", + "date": "2026-07-09", + "value": "Runs on Salesforce-managed infrastructure with published trust/status transparency and enterprise SLAs" + } + ], + "methodology": "Reliability analysis", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Salesforce Hosted MCP Servers - Standard Servers Reference", + "url": "https://developer.salesforce.com/docs/platform/hosted-mcp-servers/guide/servers-reference.html", + "date": "2026-07-09", + "value": "Standard servers for Agentforce 360 Platform, Tableau Next, and Data 360 SQL plus custom servers over flows, Apex actions, and Named Query APIs" + } + ], + "methodology": "Feature coverage assessment", + "last_verified": "2026-07-09" + }, + "deployment_flexibility": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "npm - @salesforce/mcp", + "url": "https://www.npmjs.com/package/@salesforce/mcp", + "date": "2026-06-23", + "value": "Three deployment shapes: hosted servers (GA, Enterprise Edition+ and Developer Edition at no extra cost), local DX MCP server on npm for developer workflows, and the Data 360 MCP server in developer preview (stdio, Java 17+) slated to join the hosted lineup at GA" + } + ], + "methodology": "Deployment option and edition availability review", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Salesforce-managed hosted endpoints: no infrastructure to run, scales like the REST APIs (GA 2026-04-29)", + "Org permission model enforced end-to-end: CRUD, field-level security, and sharing rules apply, with every transaction running as the authenticated user", + "OAuth with PKCE plus a dedicated MCP OAuth scope separates agent access from existing API integrations", + "Secure by default: servers disabled until explicitly enabled in Setup, with full audit trails", + "Included at no additional cost for Enterprise Edition+ orgs and Developer Edition", + "Custom servers expose flows, Apex actions, and Named Query APIs as curated tool sets; read-only sobject-reads server available", + "Open-source variants: DX MCP server (@salesforce/mcp, Apache-2.0) for dev workflows and Data 360 MCP server in developer preview" + ], + "limitations": [ + "CRM PII at scale (contacts, leads, accounts, opportunities) flows into the LLM provider's context", + "Write-capable tools (record CRUD, flows, Apex actions) act with the user's full org permissions - over-permissioned users mean over-permissioned agents", + "Requires Enterprise Edition or above (or Developer Edition); not available on lower editions", + "Admin-driven setup (Setup enablement + External Client App) adds friction versus one-click consumer MCP servers", + "MCP tool calls consume org API request limits", + "Hosted servers are closed source; open-source DX and Data 360 variants cover different scopes (dev tooling, Data 360 APIs) rather than the hosted CRM servers", + "Data 360 MCP server is developer preview only (stdio, Java 17+) and not yet part of the hosted GA lineup" + ], + "metadata": { + "license": "Proprietary (hosted service); DX MCP server Apache-2.0; Data 360 MCP server open source (developer preview)", + "supported_platforms": [ + "Hosted (Salesforce-managed endpoints, Enterprise Edition+ and Developer Edition)", + "Local DX server via npm (@salesforce/mcp)", + "Data 360 developer preview (Java 17+, stdio)" + ], + "programming_languages": [ + "N/A (hosted service)", + "TypeScript (DX server)", + "Java (Data 360 preview)" + ], + "mcp_version": "1.0", + "docs": "https://developer.salesforce.com/docs/platform/hosted-mcp-servers/guide/servers-reference.html", + "api_dependency": "Salesforce Platform APIs, Named Query APIs, Flows, Apex; Data 360 APIs (preview)", + "authentication": "OAuth 2.0 with PKCE via External Client App; dedicated MCP OAuth scope", + "package_name": "@salesforce/mcp", + "github_repo": "https://github.com/forcedotcom/d360-mcp-server", + "first_release": "2025-10 (hosted beta); 2026-04-29 (hosted GA)", + "maintained_by": "Salesforce", + "status": "Active - hosted servers GA; DX server actively released (v0.30.14, 2026-06-23); Data 360 server developer preview", + "transport_types": [ + "streamable-http (hosted)", + "stdio (DX and Data 360 servers)" + ] + }, + "use_case_ratings": { + "code-generation": { + "overall": 82, + "notes": "DX MCP server is purpose-built for Salesforce development workflows (orgs, metadata, testing)" + }, + "customer-support": { + "overall": 86, + "notes": "Strong fit: case, contact, and account context with org permissions enforced" + }, + "content-creation": { + "overall": 68, + "notes": "Useful for CRM-grounded outreach drafting; not a content platform" + }, + "data-analysis": { + "overall": 85, + "notes": "Data 360 SQL and Tableau Next servers plus Named Queries enable governed analytics" + }, + "research-assistant": { + "overall": 72, + "notes": "Good for account research within the org; limited outside CRM data" + }, + "legal-compliance": { + "overall": 62, + "notes": "Strong access controls and audit trails, but CRM PII reaches the LLM provider" + }, + "healthcare": { + "overall": 58, + "notes": "Health Cloud data routed through agent contexts raises PHI exposure concerns" + }, + "financial-analysis": { + "overall": 72, + "notes": "Governed access to revenue/opportunity data; FLS can shield sensitive fields" + }, + "education": { + "overall": 66, + "notes": "Applicable for Education Cloud orgs; standard CRM caveats apply" + }, + "creative-writing": { + "overall": 40, + "notes": "Not a fit beyond CRM-personalized copy" + } + }, + "best_for": [ + "Enterprises wanting AI agents on CRM data without building or hosting integration middleware", + "Admin-governed orgs that need CRUD/FLS/sharing enforcement and audit trails on every agent action", + "Sales and service teams working from Slack, Claude, or ChatGPT against live Salesforce data", + "Salesforce developers using the DX MCP server for org and metadata workflows" + ], + "related_entities": [ + "mcp-server-slack", + "mcp-server-stripe", + "mcp-server-zapier", + "mcp-server-atlassian" + ], + "tags": [ + "crm", + "enterprise", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ] +} diff --git a/data/mcps/mcp-server-shopify.json b/data/mcps/mcp-server-shopify.json new file mode 100644 index 0000000..59120eb --- /dev/null +++ b/data/mcps/mcp-server-shopify.json @@ -0,0 +1,452 @@ +{ + "id": "mcp-server-shopify", + "type": "mcp", + "name": "Shopify MCP Servers", + "provider": "Shopify", + "version": "2026.7", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Shopify's official MCP surface spans three layers: the Storefront MCP, live by default on every eligible store at {shop}.myshopify.com/api/mcp (public, no auth; catalog, cart, and policy tools), the local @shopify/dev-mcp stdio server (v1.14.x) for docs search and GraphQL schema work, and the open-source Shopify AI Toolkit (April 2026) adding authenticated admin operations via the Shopify CLI. Default-on endpoints across millions of stores form a huge aggregate agent surface.", + "website": "https://shopify.dev/docs/apps/build/storefront-mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 82, + "criteria": { + "api_reliability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Storefront MCP is served from Shopify's production storefront infrastructure at https://{shop}.myshopify.com/api/mcp (JSON-RPC 2.0 over HTTP POST), the same platform that serves live storefront traffic" + } + ], + "methodology": "API stability analysis of the hosted endpoint riding on Shopify's storefront-serving infrastructure", + "last_verified": "2026-07-09" + }, + "operation_success_rate": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Six documented tools (search_catalog, lookup_catalog, get_product on the UCP endpoint; get_cart, update_cart, search_shop_policies_and_faqs) map to stable storefront primitives; lookup accepts up to 10 IDs per call" + } + ], + "methodology": "Operation success testing across the documented storefront tool set against live stores", + "last_verified": "2026-07-09" + }, + "search_accuracy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Dev MCP / AI Toolkit Documentation", + "url": "https://shopify.dev/docs/apps/build/devmcp", + "date": "2026-07-09", + "value": "Dev MCP bundles Shopify documentation search, GraphQL schema introspection (introspect_graphql_schema across Admin, Storefront, Partner, Customer, and Function APIs), and code validation; storefront catalog search returns relevance-ranked product results" + } + ], + "methodology": "Relevance assessment of catalog search and docs/schema search results for representative queries", + "last_verified": "2026-07-09" + }, + "rate_limit_handling": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Storefront MCP inherits storefront-tier rate limiting; limits vary by plan (Plus stores receive higher limits and earlier access to advanced UCP capabilities) and some stores may restrict agent access entirely" + } + ], + "methodology": "Rate limiting behavior observation under sustained anonymous tool-call load across plan tiers", + "last_verified": "2026-07-09" + }, + "error_recovery": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "JSON-RPC 2.0 structured errors for malformed requests and restricted stores; Dev MCP validate tools return per-code-block validation results explaining why GraphQL was valid or invalid" + } + ], + "methodology": "Error handling testing with invalid product IDs, restricted stores, and malformed cart operations", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 66, + "criteria": { + "authentication_security": { + "score": 62, + "confidence": "high", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Storefront MCP servers don't require authentication: requests need only a Content-Type header and the store domain; any client on the internet can call the endpoint. Admin operations via the AI Toolkit use the authenticated Shopify CLI session instead" + } + ], + "methodology": "Authentication mechanism review across the anonymous storefront endpoint, local Dev MCP, and CLI-authenticated admin path", + "last_verified": "2026-07-09" + }, + "scope_limitation": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "About Storefront MCP", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp", + "date": "2026-07-09", + "value": "The public endpoint is scoped to already-public storefront data (catalog, policies, FAQs) plus cart state; admin mutations are fully separated behind the Shopify CLI authenticated path and are never reachable through /api/mcp" + } + ], + "methodology": "Permission boundary testing between the anonymous storefront surface and authenticated admin operations", + "last_verified": "2026-07-09" + }, + "prompt_injection_risk": { + "score": 48, + "confidence": "high", + "evidence": [ + { + "source": "CodeIntegrity: Shopify MCP Prompt Injection Exploit", + "url": "https://www.codeintegrity.ai/blog/shopify", + "date": "2026-07-09", + "value": "Documented indirect prompt injection: a malicious prompt hidden in a product description forced an unauthorized search_shop_catalog call and made the consuming LLM present attacker-chosen products with fabricated positive bias; merchant-authored content is untrusted input to every connected agent" + }, + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Because the endpoint is default-on across millions of stores, any store's product descriptions, policies, and FAQ text flow into agent context at ecosystem scale with no server-side sanitization guarantees" + } + ], + "methodology": "Threat modeling of merchant-authored content as an injection vector, including a publicly documented exploit against the storefront tool chain", + "last_verified": "2026-07-09" + }, + "token_exposure_risk": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Dev MCP / AI Toolkit Documentation", + "url": "https://shopify.dev/docs/apps/build/devmcp", + "date": "2026-07-09", + "value": "Storefront MCP needs no credentials at all; the local Dev MCP server runs over stdio without authentication; admin operations reuse the Shopify CLI's existing local session rather than static API keys pasted into client config" + } + ], + "methodology": "Token storage and exposure-surface analysis across the three access paths", + "last_verified": "2026-07-09" + }, + "unauthorized_action_risk": { + "score": 62, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify AI Toolkit Repository", + "url": "https://github.com/Shopify/shopify-ai-toolkit", + "date": "2026-07-09", + "value": "The AI Toolkit lets agents execute real store operations (products, inventory, store management) through the authenticated CLI path with no server-side confirmation step; on the storefront side, anonymous update_cart writes are possible but individually low-consequence" + } + ], + "methodology": "Authorization boundary testing of cart writes and CLI-authenticated admin mutations", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 73, + "criteria": { + "customer_data_exposure": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "About Storefront MCP", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp", + "date": "2026-07-09", + "value": "The storefront surface exposes only already-public catalog, policy, and FAQ data plus session cart contents; customer PII and order history are not reachable through the anonymous endpoint" + } + ], + "methodology": "Data classification of every field reachable through the anonymous storefront tool set", + "last_verified": "2026-07-09" + }, + "sensitive_data_protection": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Dev MCP npm package", + "url": "https://www.npmjs.com/package/@shopify/dev-mcp", + "date": "2026-07-09", + "value": "Dev MCP makes instrumentation calls back to Shopify by default; disabling requires setting OPT_OUT_INSTRUMENTATION=true in the client env. Admin-path results (inventory, unpublished products) flow into the connected LLM's context" + } + ], + "methodology": "Review of telemetry defaults and of sensitive data classes reachable via the authenticated admin path", + "last_verified": "2026-07-09" + }, + "organization_data_control": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "The endpoint is on by default for eligible stores (default-on for eligible US merchants since March 24, 2026); merchants can restrict agent access, but exposure is opt-out rather than opt-in and many merchants receive agent traffic without having configured anything" + } + ], + "methodology": "Review of merchant-side controls over default-on agent exposure and opt-out mechanics", + "last_verified": "2026-07-09" + }, + "third_party_data_sharing": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Privacy Policy", + "url": "https://www.shopify.com/legal/privacy", + "date": "2026-07-09", + "value": "Store content retrieved via MCP flows to whichever LLM provider the consuming agent uses (ChatGPT, Copilot, Gemini, and others surface Shopify stores by default), governed by those providers' data policies rather than Shopify's" + } + ], + "methodology": "Analysis of downstream data sharing once storefront and admin content leaves the Shopify boundary", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 77, + "criteria": { + "documentation_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Shopify Agentic Commerce Documentation", + "url": "https://shopify.dev/docs/agents", + "date": "2026-07-09", + "value": "Dedicated documentation hub covering Storefront MCP, catalog/UCP tools, Dev MCP, and the AI Toolkit, with per-tool references, request schemas, and client setup guides" + } + ], + "methodology": "Documentation completeness and accuracy review across the three server surfaces", + "last_verified": "2026-07-09" + }, + "operation_visibility": { + "score": 65, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Anonymous storefront MCP calls are not attributable to an identity and merchants get no dedicated MCP audit log; admin-path operations are visible as normal API activity, and Dev MCP calls appear only in client logs" + } + ], + "methodology": "Logging and traceability assessment across anonymous, local, and CLI-authenticated paths", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 72, + "confidence": "high", + "evidence": [ + { + "source": "Shopify AI Toolkit Repository", + "url": "https://github.com/Shopify/shopify-ai-toolkit", + "date": "2026-07-09", + "value": "AI Toolkit open-sourced under MIT on April 9, 2026 (439 stars, actively pushed through June 2026) and @shopify/dev-mcp v1.14.2 is published on npm (ISC); the hosted Storefront MCP implementation itself remains closed source" + } + ], + "methodology": "Source availability review across the open-source developer tooling and the closed hosted endpoint", + "last_verified": "2026-07-09" + }, + "api_coverage_clarity": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Tool surface explicitly enumerated: search_catalog/lookup_catalog/get_product at /api/ucp/mcp and get_cart/update_cart/search_shop_policies_and_faqs at /api/mcp, with input constraints (e.g., 10 IDs per lookup) and agent-profile requirements documented" + } + ], + "methodology": "Comparison of documented tool surface against observed endpoint behavior", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 86, + "criteria": { + "ease_of_setup": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Zero merchant setup: the endpoint is live by default on every eligible store; consumers just point a client at https://{shop}.myshopify.com/api/mcp. Dev MCP is a one-line npx @shopify/dev-mcp install" + } + ], + "methodology": "Setup complexity assessment for merchants, agent builders, and developers", + "last_verified": "2026-07-09" + }, + "api_performance": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Storefront MCP Server Documentation", + "url": "https://shopify.dev/docs/apps/build/storefront-mcp/servers/storefront", + "date": "2026-07-09", + "value": "Tools are thin JSON-RPC wrappers over Shopify's storefront-serving stack, which is heavily optimized for low-latency commerce traffic" + } + ], + "methodology": "Latency observation across catalog search, cart, and policy tools", + "last_verified": "2026-07-09" + }, + "reliability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Shopify Status", + "url": "https://www.shopifystatus.com/", + "date": "2026-07-09", + "value": "Storefront MCP rides Shopify's production storefront infrastructure with strong historical availability; the surface reached default-on general availability in Q1 2026 after a 2025 rollout" + } + ], + "methodology": "Uptime analysis of Shopify storefront infrastructure hosting the endpoint", + "last_verified": "2026-07-09" + }, + "feature_coverage": { + "score": 80, + "confidence": "high", + "evidence": [ + { + "source": "Shopify Dev MCP / AI Toolkit Documentation", + "url": "https://shopify.dev/docs/apps/build/devmcp", + "date": "2026-07-09", + "value": "Combined surface spans buyer-side commerce (catalog, cart, policies), developer tooling (docs, schema introspection, validation across Admin/Storefront/Partner/Customer/Function APIs), and CLI-backed store operations; checkout completion still hands off to Shopify-hosted flows" + } + ], + "methodology": "Feature completeness assessment against end-to-end agentic commerce and development workflows", + "last_verified": "2026-07-09" + }, + "community_adoption": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Shopify AI Toolkit Repository", + "url": "https://github.com/Shopify/shopify-ai-toolkit", + "date": "2026-07-09", + "value": "Default-on across millions of stores (5.6M stores made agent-discoverable in ChatGPT, Copilot, Google AI Mode, and Gemini as of March 2026); AI Toolkit at 439 stars within three months of its April 9, 2026 open-source launch with plugins for Claude Code, Codex, Cursor, and VS Code" + } + ], + "methodology": "Adoption analysis across merchant footprint, agent platforms, and developer tooling", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Zero-setup, default-on storefront endpoint makes every eligible store agent-ready", + "Clean separation between the anonymous public surface and CLI-authenticated admin operations", + "No credentials required for the storefront path, eliminating token-leak risk there entirely", + "Dev MCP provides docs search, GraphQL schema introspection, and validation across all major Shopify APIs", + "AI Toolkit is MIT-licensed open source with first-party plugins for major AI coding tools", + "Backed by Shopify's production commerce infrastructure and status transparency", + "UCP catalog tools position stores for cross-platform agentic commerce discovery" + ], + "limitations": [ + "Default-on public endpoints at millions of stores create an enormous aggregate attack and scraping surface that merchants did not individually opt into", + "Merchant-authored product descriptions are untrusted input; an indirect prompt injection exploit against the storefront tool chain has been publicly documented", + "Anonymous access means no identity, revocation, or per-consumer rate accountability on the storefront path", + "No dedicated merchant-facing MCP audit log for agent traffic", + "Dev MCP sends instrumentation telemetry by default (opt-out via OPT_OUT_INSTRUMENTATION)", + "Admin operations via the AI Toolkit lack a server-side confirmation step", + "The hosted Storefront MCP implementation is closed source" + ], + "metadata": { + "repository": "https://github.com/Shopify/shopify-ai-toolkit", + "package_name": "@shopify/dev-mcp", + "package_version": "1.14.2", + "license": "MIT (AI Toolkit); ISC (@shopify/dev-mcp npm package); hosted Storefront MCP closed source", + "maintained_by": "Shopify", + "github_stars": 439, + "remote_endpoint": "https://{shop}.myshopify.com/api/mcp (commerce tools); https://{shop}.myshopify.com/api/ucp/mcp (catalog tools)", + "authentication": "None (Storefront MCP, public by design); none needed (local Dev MCP); Shopify CLI session (admin operations via AI Toolkit)", + "transport_types": [ + "streamable-http / JSON-RPC 2.0 over HTTP POST (storefront, hosted)", + "stdio (@shopify/dev-mcp, local)" + ], + "installation_methods": [ + "None required (storefront endpoint live by default)", + "npx @shopify/dev-mcp", + "Shopify AI Toolkit plugin for Claude Code / Codex / Cursor / VS Code" + ], + "default_on_since": "Q1 2026 (Agentic Storefronts default-on for eligible US merchants March 24, 2026)", + "ai_toolkit_open_sourced": "2026-04-09", + "mcp_version": "1.0" + }, + "use_case_ratings": { + "customer-support": { + "overall": 84, + "notes": "Strong for shopping assistants answering catalog, policy, and FAQ questions and managing carts on any store" + }, + "code-generation": { + "overall": 88, + "notes": "Dev MCP's docs search, schema introspection, and validation make it excellent for building Shopify apps and themes with AI assistance" + }, + "data-analysis": { + "overall": 68, + "notes": "Useful for catalog and pricing lookups across stores; not designed for bulk analytics or order data" + }, + "research-assistant": { + "overall": 74, + "notes": "Good for product research and price comparison across the public storefront surface" + }, + "financial-analysis": { + "overall": 55, + "notes": "No revenue, order, or payout data is exposed through the MCP surfaces; limited to catalog-level pricing" + } + }, + "best_for": [ + "Agent builders creating shopping assistants over any Shopify store's default-on endpoint", + "Developers building Shopify apps, themes, and Functions with Dev MCP docs and schema tools", + "Merchants and operators automating store management through the AI Toolkit's CLI-authenticated path", + "Agentic commerce platforms integrating catalog discovery via the UCP tools" + ], + "not_recommended_for": [ + "Workflows needing customer PII or order history (not exposed via MCP)", + "Merchants unwilling to accept default-on public agent traffic without hardening product content", + "High-assurance environments that require per-consumer authentication and audit on every tool call" + ], + "related_entities": [ + "mcp-server-stripe", + "mcp-server-vercel", + "mcp-server-github", + "mcp-server-zapier" + ], + "tags": [ + "ecommerce", + "agentic-commerce", + "storefront", + "mcp", + "model-context-protocol" + ] +} diff --git a/data/mcps/mcp-server-snowflake.json b/data/mcps/mcp-server-snowflake.json new file mode 100644 index 0000000..3703f4c --- /dev/null +++ b/data/mcps/mcp-server-snowflake.json @@ -0,0 +1,445 @@ +{ + "id": "mcp-server-snowflake", + "type": "mcp", + "name": "MCP Snowflake Server", + "provider": "Snowflake (Official)", + "version": "Managed MCP server (GA 2025-11-04); snowflake-labs-mcp 1.x (deprecated)", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Official Snowflake MCP server, GA since 2025-11-04 as a fully managed in-account endpoint (CREATE MCP SERVER object). Exposes Cortex Analyst NL-to-SQL over semantic views, Cortex Search, Cortex Agents, SQL execution with configurable read-only mode, and custom UDF/procedure tools. OAuth 2.0 or Programmatic Access Token auth with per-tool RBAC; masking and data policies apply. The self-hosted snowflake-labs-mcp (PyPI) was deprecated in May 2026 in favor of the managed server.", + "website": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "trust_vector": { + "performance_reliability": { + "overall_score": 84, + "criteria": { + "api_reliability": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server GA release note", + "url": "https://docs.snowflake.com/en/release-notes/2025/other/2025-11-04-cortex-agents-mcp", + "date": "2025-11-04", + "value": "Managed MCP server reached general availability on 2025-11-04 after an October 2025 preview; runs inside Snowflake's platform with no separate infrastructure, inheriting platform reliability (not supported in government regions)" + } + ], + "methodology": "Endpoint availability and GA-status review", + "last_verified": "2026-07-09" + }, + "nl_to_sql_accuracy": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "Cortex Analyst tools translate natural language to SQL, constrained to semantic views (semantic models are not supported), which bounds hallucinated joins but limits coverage" + } + ], + "methodology": "NL-to-SQL capability review; semantic-view grounding improves accuracy but accuracy still depends on semantic view quality", + "last_verified": "2026-07-09" + }, + "search_retrieval_quality": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Snowflake Managed MCP Servers announcement", + "url": "https://www.snowflake.com/en/blog/managed-mcp-servers-secure-data-agents/", + "date": "2025-10-02", + "value": "Cortex Search provides semantic search over unstructured documents as MCP tools, plus Cortex Knowledge Extensions for licensed third-party content" + } + ], + "methodology": "Retrieval capability assessment based on Cortex Search service documentation", + "last_verified": "2026-07-09" + }, + "sql_execution": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "SQL execution tool supports configurable read-only mode, query timeout, and warehouse selection; custom tools via UDFs and stored procedures" + } + ], + "methodology": "SQL tool feature review", + "last_verified": "2026-07-09" + }, + "error_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "Documented guardrails: 50 tools per server maximum, 250 KB response truncation for generic/SQL tools, and a recursion depth limit of 10 invocations" + } + ], + "methodology": "Failure-mode and limit documentation review", + "last_verified": "2026-07-09" + } + } + }, + "security": { + "overall_score": 82, + "criteria": { + "authentication_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "OAuth 2.0 via Snowflake security integrations is the recommended path; Programmatic Access Tokens supported with least-privilege role assignment; Snowflake advises OAuth over hardcoded tokens and network policies for client provider IPs" + } + ], + "methodology": "Authentication mechanism review", + "last_verified": "2026-07-09" + }, + "access_control": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "Per-tool RBAC: access to the MCP server object alone does not grant tool access — individual privileges are required per tool; standard Snowflake role-based access, masking, and data policies apply to all agent operations" + } + ], + "methodology": "Access control model review; per-tool privilege separation is stronger than most MCP servers evaluated", + "last_verified": "2026-07-09" + }, + "data_modification_control": { + "score": 76, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "SQL execution tool offers a configurable read-only mode; when writes are enabled the agent can modify data within the session role's privileges, and OAuth sessions use only the user's DEFAULT_ROLE (secondary roles unsupported)" + } + ], + "methodology": "Write-surface assessment; read-only mode plus RBAC scoping mitigates but writes to production data remain possible when enabled", + "last_verified": "2026-07-09" + }, + "prompt_injection_resilience": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Snowflake Managed MCP Servers announcement", + "url": "https://www.snowflake.com/en/blog/managed-mcp-servers-secure-data-agents/", + "date": "2025-10-02", + "value": "All operations occur within Snowflake's governed perimeter with masking and policies enforced server-side, limiting blast radius; however query results containing untrusted content still reach the LLM and no result-wrapping mitigation is documented" + } + ], + "methodology": "Prompt-injection exposure analysis; governed perimeter constrains impact but the generic MCP injection vector via returned data is unmitigated", + "last_verified": "2026-07-09" + }, + "credential_management": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "Managed OAuth flow avoids long-lived credentials in client config; PATs are revocable and role-scoped; no connection strings or account passwords are placed in MCP client configuration" + } + ], + "methodology": "Credential handling review", + "last_verified": "2026-07-09" + } + } + }, + "privacy_compliance": { + "overall_score": 80, + "criteria": { + "data_residency": { + "score": 82, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "MCP server runs inside the customer's Snowflake account and region; data does not leave Snowflake's perimeter for tool execution (not available in government regions)" + } + ], + "methodology": "Data residency review", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake Trust Center", + "url": "https://www.snowflake.com/en/product/features/security-compliance/", + "date": "2026-07-09", + "value": "Underlying platform holds SOC 2 Type II, ISO 27001, PCI DSS, HIPAA, FedRAMP and other certifications that extend to in-perimeter MCP execution" + } + ], + "methodology": "Platform certification review", + "last_verified": "2026-07-09" + }, + "data_exposure_to_llm": { + "score": 68, + "confidence": "high", + "evidence": [ + { + "source": "MCP Architecture", + "url": "https://modelcontextprotocol.io/docs/learn/architecture", + "date": "2026-07-09", + "value": "Query results, search hits, and semantic view metadata returned by tools are sent to the connected LLM provider — masking policies apply before egress but returned data is exposed per the client provider's terms" + } + ], + "methodology": "Data flow analysis", + "last_verified": "2026-07-09" + }, + "governance_policy_enforcement": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake Managed MCP Servers announcement", + "url": "https://www.snowflake.com/en/blog/managed-mcp-servers-secure-data-agents/", + "date": "2025-10-02", + "value": "The same user and group access controls, masking, and data policies apply to agent access as to any other Snowflake access path — governance is enforced server-side, not by the MCP client" + } + ], + "methodology": "Governance enforcement review", + "last_verified": "2026-07-09" + } + } + }, + "trust_transparency": { + "overall_score": 76, + "criteria": { + "documentation_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake Documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "First-party documentation covering CREATE MCP SERVER YAML specification, tool types, auth setup, limits, and security guidance, maintained alongside the GA product" + } + ], + "methodology": "Documentation completeness review", + "last_verified": "2026-07-09" + }, + "open_source_transparency": { + "score": 60, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-Labs/mcp repository", + "url": "https://github.com/Snowflake-Labs/mcp", + "date": "2026-07-09", + "value": "The managed MCP server is closed-source; the open-source Apache 2.0 snowflake-labs-mcp (292 stars) is deprecated and no longer maintained (final PyPI release 2026-05-15), with users directed to the managed server" + } + ], + "methodology": "Source availability review; scored down because the actively supported path is proprietary and the open-source path is end-of-life", + "last_verified": "2026-07-09" + }, + "audit_logging": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake Access History", + "url": "https://docs.snowflake.com/en/user-guide/access-history", + "date": "2026-07-09", + "value": "Agent-issued SQL and tool invocations run as governed Snowflake queries, appearing in query history and access history for audit" + } + ], + "methodology": "Audit trail assessment", + "last_verified": "2026-07-09" + }, + "community_activity": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "snowflake-labs-mcp on PyPI", + "url": "https://pypi.org/project/snowflake-labs-mcp/", + "date": "2026-07-09", + "value": "Community activity migrated from the deprecated labs repo to Snowflake's official support channels; managed-server feedback flows through Snowflake support rather than public issue trackers" + } + ], + "methodology": "Community engagement analysis", + "last_verified": "2026-07-09" + } + } + }, + "operational_excellence": { + "overall_score": 83, + "criteria": { + "ease_of_setup": { + "score": 78, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake-managed MCP server documentation", + "url": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "date": "2026-07-09", + "value": "Setup requires a CREATE MCP SERVER statement with a YAML tool specification plus an OAuth security integration or PAT — more admin steps than one-click hosted servers, but no infrastructure to run" + } + ], + "methodology": "Setup complexity assessment", + "last_verified": "2026-07-09" + }, + "scalability": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake Warehouses", + "url": "https://docs.snowflake.com/en/user-guide/warehouses", + "date": "2026-07-09", + "value": "Tool execution scales on Snowflake virtual warehouses with per-tool warehouse selection; capacity is elastic and independent of the MCP endpoint" + } + ], + "methodology": "Scalability review", + "last_verified": "2026-07-09" + }, + "cost_efficiency": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Snowflake Cortex pricing", + "url": "https://www.snowflake.com/legal-files/CreditConsumptionTable.pdf", + "date": "2026-07-09", + "value": "Agent queries consume warehouse credits and Cortex Analyst/Search incur Cortex consumption charges; costs of exploratory agent traffic can be material without resource monitors" + } + ], + "methodology": "Cost analysis for agent-driven workloads", + "last_verified": "2026-07-09" + }, + "integration_ecosystem": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake Managed MCP Servers announcement", + "url": "https://www.snowflake.com/en/blog/managed-mcp-servers-secure-data-agents/", + "date": "2025-10-02", + "value": "Standards-based streamable HTTP endpoint works with Claude, Cursor, and other MCP clients; integrates the broader Cortex ecosystem including Knowledge Extensions (AP, Washington Post, NASDAQ, MSCI)" + } + ], + "methodology": "Ecosystem integration review", + "last_verified": "2026-07-09" + }, + "managed_operations": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Snowflake Managed MCP Servers announcement", + "url": "https://www.snowflake.com/en/blog/managed-mcp-servers-secure-data-agents/", + "date": "2025-10-02", + "value": "No separate infrastructure to deploy, patch, or monitor — the MCP endpoint is operated by Snowflake inside the customer account" + } + ], + "methodology": "Operational burden assessment", + "last_verified": "2026-07-09" + } + } + } + }, + "strengths": [ + "Fully managed, GA MCP endpoint inside the Snowflake account — no infrastructure to run", + "Per-tool RBAC: server access alone does not grant tool access; masking and data policies enforced server-side", + "OAuth 2.0 security integrations or revocable PATs — no passwords or connection strings in client config", + "Cortex Analyst NL-to-SQL grounded in semantic views reduces hallucinated queries", + "SQL execution tool with configurable read-only mode, query timeout, and warehouse selection", + "Full audit trail via Snowflake query history and access history" + ], + "limitations": [ + "Managed server is closed-source; the open-source snowflake-labs-mcp is deprecated (final release 2026-05-15)", + "Query and search results are exposed to the connected LLM provider", + "When the SQL tool's read-only mode is off, agents can write to production data within the role's privileges", + "OAuth sessions use only the user's DEFAULT_ROLE; secondary roles unsupported", + "Response truncation at 250 KB and 50-tool cap can constrain large workloads", + "Cortex Analyst supports semantic views only (not semantic models); not available in government regions", + "No documented result-wrapping or other prompt-injection countermeasure for untrusted data in results" + ], + "metadata": { + "license": "Proprietary (managed service); deprecated snowflake-labs-mcp was Apache 2.0", + "supported_platforms": ["Any MCP client (remote streamable HTTP)"], + "programming_languages": ["N/A (managed service); Python (deprecated labs server)"], + "mcp_version": "1.0", + "github_repo": "https://github.com/Snowflake-Labs/mcp", + "github_stars": 292, + "docs": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-agents-mcp", + "remote_endpoint": "https:///api/v2/databases/{database}/schemas/{schema}/mcp-servers/{name}", + "api_dependency": "Snowflake Cortex (Analyst, Search, Agents), Snowflake SQL engine", + "authentication": "OAuth 2.0 via Snowflake security integration (recommended); Programmatic Access Token", + "security_controls": ["per-tool RBAC", "SQL read-only mode", "masking and data policies", "network policies", "query timeout and warehouse scoping"], + "ga_date": "2025-11-04", + "first_release": "2025-10 (public preview); labs server early 2025", + "deprecated_predecessor": "snowflake-labs-mcp (PyPI), deprecated May 2026, final release 2026-05-15", + "maintained_by": "Snowflake", + "transport_types": ["streamable-http (managed)"] + }, + "use_case_ratings": { + "code-generation": { + "overall": 78, + "notes": "Good for generating SQL and Snowflake object DDL against real schemas" + }, + "customer-support": { + "overall": 72, + "notes": "Cortex Search over support content works well; guard customer PII exposure" + }, + "content-creation": { + "overall": 68, + "notes": "Limited direct fit; useful for data-backed content via Cortex Search" + }, + "data-analysis": { + "overall": 92, + "notes": "Primary use case: governed NL-to-SQL over semantic views plus warehouse-scale SQL" + }, + "research-assistant": { + "overall": 84, + "notes": "Strong retrieval over enterprise documents with Cortex Search and Knowledge Extensions" + }, + "legal-compliance": { + "overall": 74, + "notes": "Server-side masking, RBAC, and audit history support compliance workflows" + }, + "healthcare": { + "overall": 70, + "notes": "HIPAA-certified platform helps, but LLM exposure of results needs careful policy design" + }, + "financial-analysis": { + "overall": 86, + "notes": "Governed access to financial marts with masking policies; strong audit trail" + }, + "education": { + "overall": 76, + "notes": "Good for teaching governed analytics, though account setup is enterprise-oriented" + }, + "creative-writing": { + "overall": 50, + "notes": "Not a fit beyond retrieving reference data for narratives" + } + }, + "best_for": [ + "Enterprises exposing governed Snowflake data to AI agents without new infrastructure", + "Data teams wanting NL-to-SQL grounded in semantic views rather than raw schema guessing", + "Organizations requiring per-tool RBAC, masking, and full audit trails for agent access", + "Teams migrating off the deprecated snowflake-labs-mcp self-hosted server" + ], + "related_entities": ["mcp-server-databricks", "mcp-server-postgres", "mcp-server-elasticsearch", "mcp-server-mongodb"], + "tags": ["database", "data-warehouse", "sql", "nl-to-sql", "mcp", "model-context-protocol", "snowflake", "managed"] +} diff --git a/data/models/claude-sonnet-5.json b/data/models/claude-sonnet-5.json new file mode 100644 index 0000000..833d637 --- /dev/null +++ b/data/models/claude-sonnet-5.json @@ -0,0 +1,673 @@ +{ + "id": "claude-sonnet-5", + "type": "model", + "name": "Claude Sonnet 5", + "provider": "Anthropic", + "version": "5", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Anthropic's Sonnet-tier flagship (released 2026-06-30), positioned as the cheaper way to run agents: 72.7% SWE-bench Verified (vs Sonnet 4.6's 62.3%), 80.4% Terminal-Bench 2.1, 81.2% OSWorld-Verified — approaching Opus 4.8 at a fraction of the cost. Same $3/$15 standard price as Sonnet 4.6 with intro $2/$10 through 2026-08-31, 1M context, effort parameter incl. xhigh, and deliberately low cyber capability with default-on safeguards.", + "website": "https://www.anthropic.com/claude/sonnet", + + "trust_vector": { + "performance_reliability": { + "overall_score": 94, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "72.7% SWE-bench Verified vs Sonnet 4.6's 62.3%; described as the most agentic Sonnet model yet with sustained coding and debugging" + }, + { + "source": "MarkTechPost benchmark comparison", + "url": "https://www.marktechpost.com/2026/06/30/anthropic-claude-sonnet-5-vs-sonnet-4-6-vs-opus-4-8-agentic-coding-benchmarks-api-pricing-and-cost-performance-tradeoffs-compared/", + "date": "2026-06-30", + "value": "63.2% SWE-bench Pro (vs Sonnet 4.6's 58.1%, Opus 4.8's 69.2%); 80.4% Terminal-Bench 2.1 (vs Sonnet 4.6's 67.0%)" + }, + { + "source": "TechCrunch", + "url": "https://techcrunch.com/2026/06/30/anthropic-launches-claude-sonnet-5-as-a-cheaper-way-to-run-agents/", + "date": "2026-06-30", + "value": "Runs agentic coding workloads autonomously at a level that recently required larger, more expensive models" + } + ], + "methodology": "Provider-published coding benchmarks cross-checked against independent launch coverage; SWE-bench Verified, SWE-bench Pro, and Terminal-Bench 2.1", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "MarkTechPost benchmark comparison", + "url": "https://www.marktechpost.com/2026/06/30/anthropic-claude-sonnet-5-vs-sonnet-4-6-vs-opus-4-8-agentic-coding-benchmarks-api-pricing-and-cost-performance-tradeoffs-compared/", + "date": "2026-06-30", + "value": "57.4% Humanity's Last Exam (with tools) vs Sonnet 4.6's 46.8% and Opus 4.8's 57.9% — near-Opus reasoning at Sonnet price" + } + ], + "methodology": "Provider-published reasoning benchmarks; independent replication still limited nine days post-launch", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "MarkTechPost benchmark comparison", + "url": "https://www.marktechpost.com/2026/06/30/anthropic-claude-sonnet-5-vs-sonnet-4-6-vs-opus-4-8-agentic-coding-benchmarks-api-pricing-and-cost-performance-tradeoffs-compared/", + "date": "2026-06-30", + "value": "GDPval-AA v2 knowledge-work score of 1,618, marginally above Opus 4.8's 1,615; 81.2% OSWorld-Verified computer use" + }, + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Substantial improvement over Sonnet 4.6 on reasoning, tool use, coding, and knowledge work; strict improvement on BrowseComp agentic search" + } + ], + "methodology": "Knowledge-work and computer-use benchmark review from provider announcement and third-party comparisons", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Self-verifies its own output without explicit prompting; effort parameter gives repeatable quality/cost control across low/medium/high/xhigh" + } + ], + "methodology": "Review of documented behavior controls; repeated-run consistency data still limited post-launch. Note the updated tokenizer maps identical input to roughly 1.0-1.35x more tokens than Sonnet 4.6", + "last_verified": "2026-07-09" + }, + "latency_p50": { + "value": "~1.5s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-07-08", + "value": "Early measurements comparable to Sonnet 4.6 at low/medium effort; launch-window sample is small" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes; limited launch-window sample", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "~4.0s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models", + "date": "2026-07-08", + "value": "Early p95 around 4s; higher at xhigh effort with deeper reasoning" + } + ], + "methodology": "95th percentile response time across diverse workloads; limited launch-window sample", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "MarkTechPost benchmark comparison", + "url": "https://www.marktechpost.com/2026/06/30/anthropic-claude-sonnet-5-vs-sonnet-4-6-vs-opus-4-8-agentic-coding-benchmarks-api-pricing-and-cost-performance-tradeoffs-compared/", + "date": "2026-06-30", + "value": "1M token context window; API model string claude-sonnet-5" + } + ], + "methodology": "Specification review from launch documentation and coverage. Max output tokens not yet confirmed in public docs", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 97, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Status Page", + "url": "https://status.claude.com/", + "date": "2026-07-09", + "value": "Claude API uptime 99.57% (last 90 days); no Sonnet 5-specific incidents since the 2026-06-30 launch" + } + ], + "methodology": "Platform uptime from official status page; model-specific track record is only ~9 days old, so score held slightly below the platform baseline", + "last_verified": "2026-07-09" + } + }, + "notes": "Biggest Sonnet-generation jump in the registry: +10.4 points SWE-bench Verified and +13.4 Terminal-Bench 2.1 over Sonnet 4.6, closing most of the gap to Opus 4.8 (and edging it on GDPval-AA v2). Scores held slightly conservative pending independent benchmark replication — the model is nine days old." + }, + + "security": { + "overall_score": 91, + "criteria": { + "prompt_injection_resistance": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Improved resistance to prompt injection attacks vs Sonnet 4.6; generally safer to use in agentic contexts" + } + ], + "methodology": "Review of provider safety evaluations against OWASP LLM01 patterns; third-party red-team data still limited at launch", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Lower rate of undesirable behaviors than Sonnet 4.6, with improved refusal of malicious requests" + }, + { + "source": "Anthropic Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-06-30", + "value": "Constitutional AI alignment carried forward with well-calibrated refusals" + } + ], + "methodology": "Testing against adversarial prompt datasets and review of provider safety documentation", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Statement", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-06-30", + "value": "No training on user data without explicit consent; training opt-out by default for API traffic" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Cyber safeguards enabled by default (equivalent to Opus 4.7/4.8, less strict than Fable 5); deliberately not trained on cyber tasks — 0.0% success on full Firefox exploit development" + }, + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-06-30", + "value": "Released with safety evaluations under the Responsible Scaling Policy" + } + ], + "methodology": "Review of provider safety evaluations across harmful content and offensive-cyber categories", + "last_verified": "2026-07-09" + }, + "api_security": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-07-09", + "value": "API key authentication, HTTPS only, rate limiting, workspace scoping; same hardened API surface as the Sonnet 4.6/Opus 4.8 generation" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-07-09" + } + }, + "notes": "Strong safety posture with a notable design choice: offensive-cyber capability is intentionally kept much lower than the Opus line, with cyber safeguards on by default. Anthropic reports fewer undesirable agentic behaviors than Sonnet 4.6." + }, + + "privacy_compliance": { + "overall_score": 93, + "criteria": { + "data_residency": { + "value": "US, EU (customer choice)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Enterprise Documentation", + "url": "https://www.anthropic.com/claude/enterprise", + "date": "2026-06-30", + "value": "Data residency options for US and EU enterprise customers" + } + ], + "methodology": "Review of enterprise documentation and privacy policies", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Privacy Policy", + "url": "https://www.anthropic.com/legal/privacy", + "date": "2026-06-30", + "value": "Opt-out available; no training on API data by default" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "0 days (ephemeral)", + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Terms of Service", + "url": "https://www.anthropic.com/legal/terms", + "date": "2026-06-30", + "value": "API prompts and outputs not retained (except for trust & safety)" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Privacy Documentation", + "url": "https://docs.anthropic.com/en/docs/resources/data-protection", + "date": "2026-06-30", + "value": "Customer responsible for PII redaction" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Trust Center", + "url": "https://trust.anthropic.com/", + "date": "2026-07-09", + "value": "SOC 2 Type II, GDPR compliant, HIPAA eligible — same posture as the rest of the Claude line" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://docs.anthropic.com/en/docs/resources/data-protection", + "date": "2026-06-30", + "value": "Ephemeral data processing; zero data retention configuration available" + } + ], + "methodology": "Review of data handling practices", + "last_verified": "2026-07-09" + } + }, + "notes": "Inherits Anthropic's full enterprise privacy posture from day one: ephemeral data handling, SOC 2 Type II, GDPR, HIPAA eligible." + }, + + "trust_transparency": { + "overall_score": 88, + "criteria": { + "explainability": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Effort parameter (low/medium/high/xhigh) makes reasoning depth explicit and controllable; self-verification behavior surfaced in outputs" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Lower hallucination and sycophancy rates than Sonnet 4.6 per launch safety evaluations" + } + ], + "methodology": "Review of provider evaluations; independent factual-QA measurement still limited at launch", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Responsible Scaling Policy", + "url": "https://www.anthropic.com/news/anthropics-responsible-scaling-policy", + "date": "2026-06-30", + "value": "Regular bias testing and mitigation under the Responsible Scaling Policy" + } + ], + "methodology": "Evaluation on bias benchmarks and diverse demographic testing", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Lower sycophancy and self-verification behavior suggest improved calibration over Sonnet 4.6" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Detailed launch documentation covering benchmarks, safety evaluations (incl. cyber capability testing), tokenizer change, and pricing; a BrowseComp methodology correction was published transparently as a changelog entry" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Public Statements", + "url": "https://www.anthropic.com/news", + "date": "2026-06-30", + "value": "General description provided; detailed sources and knowledge cutoff not disclosed at launch" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "Constitutional AI", + "url": "https://www.anthropic.com/news/claudes-constitution", + "date": "2026-06-30", + "value": "Constitutional AI guardrails plus default-on cyber safeguards and deliberately restricted offensive-cyber training" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-07-09" + } + }, + "notes": "Transparent launch documentation, including an unusually candid cyber-capability section and a public benchmark-methodology correction. Knowledge cutoff not yet published." + }, + + "operational_excellence": { + "overall_score": 90, + "criteria": { + "api_design_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Documentation", + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "date": "2026-07-09", + "value": "Same modern API surface as the Sonnet 4.6 generation: effort parameter (low/medium/high/xhigh), structured outputs, streaming, tool use; stable claude-sonnet-5 model string" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic SDKs", + "url": "https://github.com/anthropics", + "date": "2026-06-30", + "value": "Official SDKs (Python, TypeScript, Java, Go, Ruby, C#, PHP) with day-one Sonnet 5 support" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic API Versioning", + "url": "https://platform.claude.com/docs/en/api/versioning", + "date": "2026-06-30", + "value": "Clear versioning with advance deprecation notice; Sonnet 4.6 remains Active with tentative retirement not sooner than 2027-02-17, giving a long migration runway" + } + ], + "methodology": "Review of versioning policy and historical practices", + "last_verified": "2026-07-09" + }, + "monitoring_observability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic Console", + "url": "https://console.anthropic.com/", + "date": "2026-06-30", + "value": "Usage dashboard with metrics, cost tracking, and workspace controls" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 91, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Support", + "url": "https://support.anthropic.com/", + "date": "2026-06-30", + "value": "Email support, developer community, comprehensive docs and migration guidance from Sonnet 4.6" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "Anthropic: Introducing Claude Sonnet 5", + "url": "https://www.anthropic.com/news/claude-sonnet-5", + "date": "2026-06-30", + "value": "Default model for Claude Free and Pro plans; available on Max/Team/Enterprise, Claude Code, the Claude API, Claude Platform on AWS, and Microsoft Foundry — Google Vertex AI listed as coming soon" + } + ], + "methodology": "Analysis of availability surfaces and third-party integrations; score held below Sonnet 4.6 until Vertex AI availability lands and framework defaults migrate", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Anthropic Commercial Terms", + "url": "https://www.anthropic.com/legal/commercial-terms", + "date": "2026-06-30", + "value": "Standard commercial terms; enterprise agreements available" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-07-09" + } + }, + "notes": "Drop-in adoption for existing Claude users, with one real migration caveat: the updated tokenizer maps the same input to roughly 1.0-1.35x more tokens than Sonnet 4.6, so per-request cost can rise even at identical per-token prices. Vertex AI availability still pending as of 2026-07-09." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 95, + "notes": "72.7% SWE-bench Verified and 80.4% Terminal-Bench 2.1 — near-Opus agentic coding at Sonnet pricing. The new best default for production coding agents.", + "alternatives": ["claude-opus-4-8", "claude-fable-5"] + }, + "customer-support": { + "overall": 93, + "notes": "The Sonnet support sweet spot, now smarter at the same price; use effort 'low' for high-volume tiers. Intro pricing makes migration trials cheap through August.", + "alternatives": ["claude-haiku-4-5", "gpt-5-5"] + }, + "content-creation": { + "overall": 91, + "notes": "Strong long-form and marketing content with fast turnaround; Opus/Fable tiers still lead on the most nuanced pieces.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + }, + "data-analysis": { + "overall": 92, + "notes": "1M context for large datasets with GDPval-AA v2 knowledge-work performance that edges Opus 4.8, at workhorse pricing.", + "alternatives": ["gemini-3-1-pro", "claude-opus-4-8"] + }, + "research-assistant": { + "overall": 92, + "notes": "1M context, strict BrowseComp improvement over Sonnet 4.6, and strong synthesis. Opus/Fable preferred for the hardest research.", + "alternatives": ["claude-fable-5", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 90, + "notes": "Anthropic cites legal research and insurance workflows as target use cases; HIPAA eligible with 1M context for contract repositories.", + "alternatives": ["claude-opus-4-8", "claude-fable-5"] + }, + "healthcare": { + "overall": 89, + "notes": "HIPAA eligible with strong privacy controls; real-world clinical validation is still early nine days post-launch.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-4-6"] + }, + "financial-analysis": { + "overall": 91, + "notes": "Near-Opus reasoning (57.4% HLE with tools) at predictable cost; escalate the hardest modeling to Opus 4.8 or Fable 5.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "education": { + "overall": 92, + "notes": "Fast, patient explanations at a price point that scales; default model on Free/Pro plans broadens reach.", + "alternatives": ["claude-haiku-4-5", "gpt-5-5"] + }, + "creative-writing": { + "overall": 88, + "notes": "Capable creative writing with good narrative flow; Opus tier produces more distinctive prose.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + } + }, + + "strengths": [ + "72.7% SWE-bench Verified and 80.4% Terminal-Bench 2.1 — a 10+ point generation-over-generation jump at the same standard price", + "Near-Opus performance (57.4% HLE with tools vs Opus 4.8's 57.9%; edges Opus 4.8 on GDPval-AA v2) at $3/$15", + "Introductory $2/$10 pricing through 2026-08-31 makes agent migration trials cheap", + "1M token context window with effort parameter (low/medium/high/xhigh)", + "Self-verification behavior and lower hallucination/sycophancy rates than Sonnet 4.6", + "Deliberately low offensive-cyber capability with default-on safeguards — a safety plus for most deployments", + "HIPAA eligible with ephemeral data handling from day one" + ], + + "limitations": [ + "Released 2026-06-30 — independent benchmark replication and production track record still limited", + "Updated tokenizer maps identical input to roughly 1.0-1.35x more tokens than Sonnet 4.6, partially offsetting per-token price parity", + "Google Vertex AI availability still pending at evaluation time (AWS and Microsoft Foundry live)", + "Intentionally weak at sanctioned cybersecurity work — use Opus tier for authorized cyber tasks", + "Knowledge cutoff and max output tokens not yet published", + "Opus 4.8 still leads on the hardest agentic coding (69.2% vs 63.2% SWE-bench Pro)" + ], + + "best_for": [ + "Production agent fleets seeking Opus-adjacent quality at Sonnet cost (Anthropic's own 'cheaper way to run agents' positioning)", + "Agentic coding, browser, and terminal automation at scale", + "High-volume customer support and conversational applications", + "Long-context analysis of large document sets on a budget", + "Enterprise workflows (legal research, insurance) needing strong compliance posture" + ], + + "not_recommended_for": [ + "Frontier-difficulty reasoning and coding where Opus 4.8 or Fable 5 is warranted", + "Sanctioned offensive-security / cyber work (capability deliberately restricted)", + "Teams requiring Google Vertex AI availability today", + "Audio processing applications" + ], + + "metadata": { + "pricing": { + "input": "$3.00 per 1M tokens", + "output": "$15.00 per 1M tokens", + "notes": "Introductory pricing $2/$10 per 1M through 2026-08-31, then standard $3/$15 (same as Sonnet 4.6). Prompt caching (cache reads at 0.1x input) and 50% Batch API discount apply. Note the updated tokenizer yields ~1.0-1.35x more tokens for identical input vs Sonnet 4.6.", + "last_verified": "2026-07-09" + }, + "context_window": 1000000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)", "document", "computer-use"], + "api_endpoint": "https://api.anthropic.com/v1/messages", + "api_model_id": "claude-sonnet-5", + "open_source": false, + "architecture": "Transformer-based with Constitutional AI alignment, effort parameter (low/medium/high/xhigh), self-verification, and an updated tokenizer", + "parameters": "Not disclosed", + "knowledge_cutoff": "Not published at launch (as of 2026-07-09)", + "release_date": "2026-06-30" + }, + + "related_entities": ["claude-sonnet-4-6", "claude-opus-4-8", "claude-fable-5", "claude-haiku-4-5", "gpt-5-6"], + + "tags": [ + "coding", + "agentic", + "production", + "enterprise", + "hipaa-eligible", + "effort-parameter", + "computer-use", + "long-context", + "value-workhorse", + "new-release" + ] +} diff --git a/data/models/glm-5-2.json b/data/models/glm-5-2.json new file mode 100644 index 0000000..2612467 --- /dev/null +++ b/data/models/glm-5-2.json @@ -0,0 +1,659 @@ +{ + "id": "glm-5-2", + "type": "model", + "name": "GLM-5.2", + "provider": "Z.ai (Zhipu AI)", + "version": "20260613", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Z.ai's MIT-licensed 744B-parameter MoE (40B active) launched June 2026 with a 1M-token context via IndexShare sparse attention. Leading open-weight model on Artificial Analysis Intelligence Index v4.1 (51), with 62.1 SWE-bench Pro, 81.0 Terminal-Bench 2.1, and 99.2% AIME 2026 at $1.40/$4.40 per 1M tokens. Weights published 2026-06-16.", + "website": "https://huggingface.co/zai-org/GLM-5.2", + + "trust_vector": { + "performance_reliability": { + "overall_score": 93, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "GLM-5.2 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "62.1 SWE-bench Pro (top open-weight, just behind Claude Opus 4.8), 81.0 Terminal-Bench 2.1" + }, + { + "source": "Artificial Analysis / Code Arena", + "url": "https://artificialanalysis.ai/models/glm-5-2", + "date": "2026-06-17", + "value": "2nd on Code Arena WebDev leaderboard, behind only Claude Fable 5; leads open weights on GDPval-AA v2 agentic benchmark, roughly level with GPT-5.5 xhigh" + }, + { + "source": "VentureBeat launch coverage", + "url": "https://venturebeat.com/technology/z-ais-open-weights-glm-5-2-beats-gpt-5-5-on-multiple-long-horizon-coding-benchmarks-for-1-6th-the-cost", + "date": "2026-06-13", + "value": "Beats GPT-5.5 on multiple long-horizon coding benchmarks at roughly 1/6th the cost" + } + ], + "methodology": "Vendor benchmarks corroborated by independent leaderboards (Artificial Analysis, Code Arena) within weeks of launch", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 94, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5.2 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "99.2% AIME 2026, 91.2% GPQA-Diamond, 40.5 Humanity's Last Exam (54.7 with tools)" + } + ], + "methodology": "Vendor-reported competition and graduate-level reasoning benchmarks; broad independent replication still pending given launch recency", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/glm-5-2", + "date": "2026-06-17", + "value": "Leading open-weights model on Intelligence Index v4.1 with a score of 51" + }, + { + "source": "Simon Willison review", + "url": "https://simonwillison.net/2026/jun/17/glm-52/", + "date": "2026-06-17", + "value": "\"Probably the most powerful text-only open weights LLM\"; strong independent hands-on results" + } + ], + "methodology": "Independent composite benchmarking and third-party hands-on evaluation", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Simon Willison review", + "url": "https://simonwillison.net/2026/jun/17/glm-52/", + "date": "2026-06-17", + "value": "Stable results across tasks but token-hungry (~43K output tokens per benchmark task vs 24-37K for peers); only weeks of community testing so far" + } + ], + "methodology": "Early community testing with repeated prompts; limited observation window since June 2026 launch", + "last_verified": "2026-07-09" + }, + "latency_p50": { + "value": "3.0s", + "confidence": "low", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/glm-5-2", + "date": "2026-06-17", + "value": "IndexShare keeps 1M-context inference manageable (2.9x per-token FLOP reduction), but high reasoning-token usage lengthens end-to-end completions" + } + ], + "methodology": "Early median latency observations; limited data given launch recency", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "7.0s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models/glm-5-2", + "date": "2026-06-17", + "value": "p95 ~7.0s; long-context and heavy-reasoning requests run substantially longer" + } + ], + "methodology": "95th percentile response time from early third-party measurements", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "GLM-5.2 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "1M-token context window (5x GLM-5.1's 200K) with up to 131,072 output tokens (163,840 for reasoning)" + } + ], + "methodology": "Official specification from model card", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 92, + "confidence": "low", + "evidence": [ + { + "source": "Z.ai Platform", + "url": "https://z.ai/", + "date": "2026-07-01", + "value": "First-party API stable in first weeks; open weights enable self-hosted redundancy, but no long-run availability record yet" + } + ], + "methodology": "Review of platform availability since launch; observation window under one month", + "last_verified": "2026-07-09" + } + }, + "notes": "Strongest open-weight release to date on independent measures: leads Artificial Analysis Intelligence Index v4.1 (51) and GDPval-AA v2, 62.1 SWE-bench Pro, 81.0 Terminal-Bench 2.1, 99.2% AIME 2026. Launch recency (June 2026) limits consistency, latency, and uptime confidence; the model is notably token-hungry." + }, + + "security": { + "overall_score": 79, + "criteria": { + "prompt_injection_resistance": { + "score": 79, + "confidence": "low", + "evidence": [ + { + "source": "GLM-5.2 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Safety tuning consistent with GLM-5/5.1 lineage; no dedicated third-party injection audit published for GLM-5.2 yet" + } + ], + "methodology": "Review of safety documentation and GLM-family precedent against OWASP LLM01 patterns; model too new for mature red-team coverage", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 77, + "confidence": "low", + "evidence": [ + { + "source": "Community red-teaming", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-20", + "value": "Standard alignment; open weights allow guardrail removal in derivatives; limited adversarial testing published since launch" + } + ], + "methodology": "Early testing against adversarial prompt datasets; deployer-dependent for self-hosted use", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Privacy Policy", + "url": "https://z.ai/", + "date": "2026-06-13", + "value": "Standard data handling on first-party API; full control when self-hosted" + } + ], + "methodology": "Analysis of privacy policies and self-hosting data-control options", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5.2 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Safety post-training; refusal behavior in line with GLM-5/5.1 and peer open frontier models" + } + ], + "methodology": "Safety testing across harmful content categories, anchored to GLM-family precedent", + "last_verified": "2026-07-09" + }, + "api_security": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai API Documentation", + "url": "https://docs.z.ai/", + "date": "2026-06-13", + "value": "API key authentication, HTTPS only, rate limiting; OpenAI-compatible endpoints shared with the GLM-5 family" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-07-09" + } + }, + "notes": "Inherits the GLM family's solid-but-unaudited open-model posture. No third-party security audit yet; the model is under a month old, so red-team coverage is thin. Self-hosting shifts responsibility to the deployer." + }, + + "privacy_compliance": { + "overall_score": 75, + "criteria": { + "data_residency": { + "value": "China (first-party Z.ai API); any jurisdiction when self-hosted or via Western hosts", + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Platform Documentation", + "url": "https://docs.z.ai/", + "date": "2026-06-13", + "value": "Zhipu/Z.ai is a China-based provider; first-party API data processed under Chinese jurisdiction" + }, + { + "source": "OpenRouter availability", + "url": "https://openrouter.ai/z-ai/glm-5.2", + "date": "2026-06-17", + "value": "MIT weights hosted by Western inference providers (OpenRouter, DeepInfra, Featherless), enabling non-China residency" + } + ], + "methodology": "Review of provider jurisdiction and third-party hosting options", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Privacy Policy", + "url": "https://z.ai/", + "date": "2026-06-13", + "value": "Standard API data terms unchanged from GLM-5/5.1; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "Per Z.ai policy on first-party API (China jurisdiction); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Terms of Service", + "url": "https://z.ai/", + "date": "2026-06-13", + "value": "First-party retention governed by Chinese data regulations; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service and deployment-dependent retention", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Documentation", + "url": "https://docs.z.ai/", + "date": "2026-06-13", + "value": "Customer responsible for PII redaction; no managed PII tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai public materials", + "url": "https://z.ai/", + "date": "2026-06-13", + "value": "No published SOC 2 / HIPAA / GDPR attestations for the first-party API; Western hosts may carry their own certifications" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Open weights on Hugging Face", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "MIT-licensed self-hosting gives complete data control and zero external retention; BF16 deployment needs over 1TB of GPU VRAM" + } + ], + "methodology": "Review of self-hosting deployment options enabling zero retention", + "last_verified": "2026-07-09" + } + }, + "notes": "Same posture as GLM-5: first-party API under Chinese jurisdiction is a material caveat for Western regulated industries, cleanly mitigated by the unencumbered MIT weights via self-hosting or Western hosts." + }, + + "trust_transparency": { + "overall_score": 80, + "criteria": { + "explainability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5.2 documentation", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Reasoning traces and agentic tool-call logs inspectable; up to 163,840 reasoning output tokens exposed to the caller" + } + ], + "methodology": "Evaluation of reasoning transparency and trajectory inspectability", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis GDPval-AA v2", + "url": "https://artificialanalysis.ai/models/glm-5-2", + "date": "2026-06-17", + "value": "Leads open weights on real-world agentic benchmark, indicating disciplined grounded behavior; dedicated factuality studies not yet published" + } + ], + "methodology": "Inference from grounded agentic benchmarks; limited dedicated factual-QA data given launch recency", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "GLM-5.2 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Limited published bias evaluation" + } + ], + "methodology": "Review of published bias benchmarks and community evaluations", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 77, + "confidence": "low", + "evidence": [ + { + "source": "Model behavior testing", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-20", + "value": "Expresses uncertainty adequately; no calibrated confidence outputs; limited testing since launch" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face model card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Detailed disclosure: MoE architecture, IndexShare sparse attention (2.9x per-token FLOP reduction at 1M context), improved MTP speculative decoding, full benchmark tables, deployment guides" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5.2 technical disclosure", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Architecture and training recipe outlined in the GLM technical report lineage; pretraining token count and data sources not disclosed for 5.2" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "GLM-5.2 Model Card", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Built-in safety tuning; deployers of open weights must layer their own guardrails" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-07-09" + } + }, + "notes": "Strong architectural transparency (IndexShare, MTP, benchmark tables) with rapid independent verification by Artificial Analysis and community reviewers. Training data detail is thinner than GLM-5's, and bias/safety evaluations remain unpublished. Note vendor materials cite 753B total parameters while Artificial Analysis lists 744B/40B active; active-parameter count is consistent across sources." + }, + + "operational_excellence": { + "overall_score": 83, + "criteria": { + "api_design_quality": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Z.ai API Documentation", + "url": "https://docs.z.ai/", + "date": "2026-06-13", + "value": "OpenAI-compatible API with streaming, tool calling, structured output; same interface as the rest of the GLM-5 family" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai GitHub organization", + "url": "https://github.com/zai-org", + "date": "2026-06-16", + "value": "Official repos with deployment recipes; OpenAI-compatible so mainstream SDKs work; Z.ai coding CLI with promotional free-token allowance" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "GLM release history", + "url": "https://huggingface.co/zai-org", + "date": "2026-06-16", + "value": "Third flagship in five months (GLM-5 Feb, GLM-5.1 Mar/Apr, GLM-5.2 Jun); prior weights remain available, but the cadence creates version-tracking overhead" + }, + { + "source": "GLM-5.2 launch coverage", + "url": "https://simonwillison.net/2026/jun/17/glm-52/", + "date": "2026-06-17", + "value": "API launched 2026-06-13; open weights followed 2026-06-16 as promised" + } + ], + "methodology": "Review of versioning practices and weight availability across releases", + "last_verified": "2026-07-09" + }, + "monitoring_observability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai Platform", + "url": "https://z.ai/", + "date": "2026-06-13", + "value": "Basic usage dashboard; self-hosted observability is deployer-built" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "Z.ai community channels", + "url": "https://github.com/zai-org", + "date": "2026-06-16", + "value": "Active GitHub support and documentation; limited English-language enterprise support" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Inference ecosystem", + "url": "https://openrouter.ai/z-ai/glm-5.2", + "date": "2026-06-20", + "value": "Day-one SGLang/vLLM/Transformers/KTransformers support plus Ascend NPU; already on OpenRouter, DeepInfra, and Featherless within weeks — though the hosting ecosystem is younger than GLM-5's" + } + ], + "methodology": "Analysis of third-party hosting, integrations, and tooling; conservative given under a month since weights release", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "MIT License", + "url": "https://huggingface.co/zai-org/GLM-5.2", + "date": "2026-06-16", + "value": "Unencumbered MIT license — \"no regional limits\"; unrestricted commercial use and derivatives" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-07-09" + } + }, + "notes": "Clean MIT licensing and fast third-party host adoption. The family's rapid release cadence (three flagships in five months) remains the main operational overhead; ecosystem depth for 5.2 specifically is still building given the mid-June weights release." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 95, + "notes": "Top open-weight coding model: 62.1 SWE-bench Pro and 81.0 Terminal-Bench 2.1, just behind Claude Opus 4.8, with a 400K coding context at 1/6th GPT-5.5's cost.", + "alternatives": ["kimi-k2-6", "claude-opus-4-8", "gpt-5-5"] + }, + "customer-support": { + "overall": 82, + "notes": "Capable and inexpensive, but token-hungry reasoning is wasteful for simple support flows.", + "alternatives": ["command-a-plus", "minimax-m2"] + }, + "content-creation": { + "overall": 84, + "notes": "Strong long-form generation with 1M context for whole-corpus grounding.", + "alternatives": ["claude-opus-4-8", "kimi-k2-6"] + }, + "data-analysis": { + "overall": 90, + "notes": "Near-perfect competition math (99.2% AIME 2026) and leading agentic benchmark results for analysis pipelines.", + "alternatives": ["kimi-k2-6", "deepseek-v4"] + }, + "research-assistant": { + "overall": 92, + "notes": "Leads open weights on GDPval-AA v2; 1M-token context handles entire document collections in one pass.", + "alternatives": ["kimi-k2-6", "claude-opus-4-8"] + }, + "legal-compliance": { + "overall": 70, + "notes": "China-jurisdiction first-party API and absent Western certifications are blockers unless self-hosted; 1M context is attractive for contract corpora once mitigated.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 68, + "notes": "Not recommended via first-party API; self-hosted deployment in a compliant environment is the only viable path.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 87, + "notes": "Top-tier quantitative reasoning; regulated firms should self-host or use certified Western hosts.", + "alternatives": ["command-a-plus", "kimi-k2-6"] + }, + "education": { + "overall": 89, + "notes": "Outstanding math and science tutoring (99.2% AIME, 91.2% GPQA-Diamond); pricier than GLM-5 but still budget-friendly.", + "alternatives": ["deepseek-v3-2", "glm-5"] + }, + "creative-writing": { + "overall": 81, + "notes": "Competent prose; optimized for coding and agentic work rather than creative style.", + "alternatives": ["claude-opus-4-8", "minimax-m2"] + } + }, + + "strengths": [ + "Leading open-weight model on independent measures: Artificial Analysis Intelligence Index v4.1 (51) and GDPval-AA v2", + "Top open-weight coding results: 62.1 SWE-bench Pro, 81.0 Terminal-Bench 2.1, 2nd on Code Arena WebDev", + "1M-token context (5x GLM-5.1) made affordable by IndexShare sparse attention (2.9x FLOP reduction)", + "Unencumbered MIT license with weights published three days after API launch", + "Frontier-competitive reasoning: 99.2% AIME 2026, 91.2% GPQA-Diamond, 54.7 HLE-with-tools", + "Roughly 1/6th the cost of GPT-5.5 at $1.40/$4.40 per 1M tokens" + ], + + "limitations": [ + "First-party Z.ai API processes data under Chinese jurisdiction with limited Western compliance certifications", + "Token-hungry: ~43K output tokens per benchmark task vs 24-37K for peers, inflating effective cost and latency", + "Text-only — no vision or audio modalities", + "Under a month old: consistency, uptime, and security evidence still immature", + "Limited published bias, safety, and red-team evaluations", + "Self-hosting requires over 1TB of GPU VRAM in BF16", + "Parameter count reported inconsistently (744B by Artificial Analysis vs 753B in vendor materials)" + ], + + "best_for": [ + "Repository-scale agentic coding with long-horizon, multi-file context (1M tokens)", + "Cost-sensitive teams wanting frontier-class coding at ~1/6th GPT-5.5 pricing", + "Whole-corpus research and analysis workloads exploiting the 1M context", + "Organizations wanting clean MIT-licensed weights for self-hosted data control" + ], + + "not_recommended_for": [ + "Regulated Western workloads (healthcare, legal, finance) on the first-party API", + "Multimodal applications requiring image or audio input", + "Token-budget-constrained or latency-critical applications given heavy reasoning-token usage", + "Teams without GPU infrastructure who also cannot accept China-jurisdiction processing" + ], + + "metadata": { + "pricing": { + "input": "$1.40 per 1M tokens ($0.26 cache hit)", + "output": "$4.40 per 1M tokens", + "notes": "First-party Z.ai API pricing at launch, confirmed July 2026. OpenRouter routes from ~$1.00/$4.00; DeepInfra ~$1.20/$4.10-4.20 (fp4). Pricier than GLM-5 ($0.60/$1.92) but roughly 1/6th of GPT-5.5.", + "last_verified": "2026-07-09" + }, + "context_window": 1000000, + "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], + "modalities": ["text"], + "api_endpoint": "https://api.z.ai/api/paas/v4/chat/completions", + "open_source": true, + "license": "MIT", + "architecture": "Mixture-of-Experts: 744B total (vendor cites 753B) / 40B active parameters, IndexShare sparse attention (indexer shared across every four sparse-attention layers), improved MTP speculative decoding", + "parameters": "744B total / 40B active", + "release_date": "2026-06-13" + }, + + "related_entities": ["glm-5", "kimi-k2-6", "deepseek-v4", "minimax-m2", "claude-opus-4-8"], + + "tags": [ + "coding", + "reasoning", + "open-source", + "mit-license", + "mixture-of-experts", + "agentic", + "long-context", + "cost-effective", + "chinese-provider", + "self-hostable" + ] +} diff --git a/data/models/gpt-5-6.json b/data/models/gpt-5-6.json new file mode 100644 index 0000000..04bc6db --- /dev/null +++ b/data/models/gpt-5-6.json @@ -0,0 +1,661 @@ +{ + "id": "gpt-5-6", + "type": "model", + "name": "GPT-5.6", + "provider": "OpenAI", + "version": "gpt-5-6-2026-07-09", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "OpenAI's GPT-5.6 family — Sol (flagship, $5/$30), Terra (balanced, $2.50/$15), Luna (fast, $1/$6 per 1M) — previewed 2026-06-26 under US-government-requested partner-only restrictions and publicly released 2026-07-09. Sol posts 88.8% Terminal-Bench 2.1 (91.9% in Ultra mode); Terra is reported GPT-5.5-class at half the price. ~1.5M context reported but unconfirmed. Launch-day evaluation: independent verification is still very limited.", + "website": "https://openai.com/index/previewing-gpt-5-6-sol/", + + "trust_vector": { + "performance_reliability": { + "overall_score": 94, + "criteria": { + "task_accuracy_code": { + "score": 96, + "confidence": "medium", + "evidence": [ + { + "source": "ExplainX GPT-5.6 guide", + "url": "https://explainx.ai/blog/gpt-5-6-release-date-features-benchmarks-2026", + "date": "2026-07-09", + "value": "Terminal-Bench 2.1: Sol 88.8% (91.9% in Ultra mode with subagents), Terra 82.5%, Luna 84.3% — vs GPT-5.5's 88.0% baseline" + }, + { + "source": "OpenAI: Previewing GPT-5.6 Sol", + "url": "https://openai.com/index/previewing-gpt-5-6-sol/", + "date": "2026-06-26", + "value": "Sol positioned for extended coding sessions and advanced agent-driven workflows; preview ran in API and Codex for trusted partners" + } + ], + "methodology": "Provider-reported benchmarks from the preview announcement and launch coverage; no independent SWE-bench-style replication exists yet on public release day", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 95, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI: Previewing GPT-5.6 Sol", + "url": "https://openai.com/index/previewing-gpt-5-6-sol/", + "date": "2026-06-26", + "value": "Sol described as built for the most demanding complex-reasoning and security-focused tasks, with a max reasoning effort mode and Ultra subagent mode" + } + ], + "methodology": "Provider positioning and preview-partner reports; quantitative reasoning benchmarks (GPQA, ARC-AGI-2 class) not yet independently published for the family", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 94, + "confidence": "low", + "evidence": [ + { + "source": "VentureBeat", + "url": "https://venturebeat.com/technology/openai-unveils-gpt-5-6-sol-terra-and-luna-models-but-only-accessible-to-limited-preview-partners-for-now-per-us-gov", + "date": "2026-06-26", + "value": "Three-tier family: Sol flagship, Terra balanced for everyday work, Luna fast and affordable; preview initially limited to ~20 vetted partners" + }, + { + "source": "ExplainX GPT-5.6 guide", + "url": "https://explainx.ai/blog/gpt-5-6-release-date-features-benchmarks-2026", + "date": "2026-07-09", + "value": "Terra reported to deliver GPT-5.5-class capability at half the price ($2.50/$15 vs $5/$30)" + } + ], + "methodology": "Launch coverage review; the Terra-equals-GPT-5.5 claim is provider/partner-reported and not yet independently verified", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 92, + "confidence": "low", + "evidence": [ + { + "source": "ExplainX GPT-5.6 guide", + "url": "https://explainx.ai/blog/gpt-5-6-release-date-features-benchmarks-2026", + "date": "2026-07-09", + "value": "Preview partners report improved token efficiency — fewer tokens to accomplish the same long-horizon work" + } + ], + "methodology": "Preview-partner reports only; no public repeated-run consistency data on launch day", + "last_verified": "2026-07-09" + }, + "latency_p50": { + "value": "~1.0s standard; up to 750 tokens/s for Sol on Cerebras (July rollout)", + "confidence": "low", + "evidence": [ + { + "source": "OpenAI Developer Community announcement", + "url": "https://community.openai.com/t/introducing-gpt-5-6-series-sol-terra-and-luna/1384931", + "date": "2026-07-08", + "value": "GPT-5.6 Sol launching on Cerebras at up to 750 tokens per second in July" + } + ], + "methodology": "Provider/partner statements; no public latency distribution data exists on release day", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "Unknown — no public data on release day", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/", + "date": "2026-07-09", + "value": "Independent latency benchmarking not yet published for the GPT-5.6 family as of public release day" + } + ], + "methodology": "95th percentile latency requires post-launch measurement; will vary with reasoning effort and Ultra mode", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "~1,500,000 tokens (reported, unconfirmed)", + "confidence": "low", + "evidence": [ + { + "source": "ExplainX GPT-5.6 guide", + "url": "https://explainx.ai/blog/gpt-5-6-release-date-features-benchmarks-2026", + "date": "2026-07-09", + "value": "Context window of up to 1.5M tokens widely reported but not confirmed in OpenAI's official June 26 preview post" + } + ], + "methodology": "Secondary-source reports; official platform documentation had not published a definitive figure at evaluation time", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 90, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI Status", + "url": "https://status.openai.com/", + "date": "2026-07-09", + "value": "OpenAI platform 99.9% uptime (last 90 days); GPT-5.6 public endpoints went live 2026-07-09 with no model-specific track record" + } + ], + "methodology": "Platform uptime baseline; model-specific score held down because public availability began today", + "last_verified": "2026-07-09" + } + }, + "notes": "Launch-day evaluation (public release 2026-07-09). Provider-reported numbers are strong — Sol beats GPT-5.5 on Terminal-Bench 2.1 (88.8% vs 88.0%, 91.9% Ultra) — but nearly everything else, including the ~1.5M context and the Terra-equals-GPT-5.5-at-half-price claim, awaits independent verification. Note the oddity that Luna (84.3%) is reported above Terra (82.5%) on Terminal-Bench 2.1; treat tier orderings as provisional." + }, + + "security": { + "overall_score": 88, + "criteria": { + "prompt_injection_resistance": { + "score": 89, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-06-26", + "value": "Multi-layer injection defenses carried forward from GPT-5.5; family-specific red-team results not yet public" + } + ], + "methodology": "Inherited safety-stack review; no third-party OWASP LLM01 testing published for GPT-5.6 at launch", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "VentureBeat", + "url": "https://venturebeat.com/technology/openai-unveils-gpt-5-6-sol-terra-and-luna-models-but-only-accessible-to-limited-preview-partners-for-now-per-us-gov", + "date": "2026-06-26", + "value": "US government requested a vetted-partner-only start; the family underwent a two-week government-linked partner evaluation period before public release" + }, + { + "source": "ExplainX GPT-5.6 guide", + "url": "https://explainx.ai/blog/gpt-5-6-release-date-features-benchmarks-2026", + "date": "2026-07-09", + "value": "Preview deployed for government vetted-partner evaluation under the Cyber EO framework (deadline 2026-08-01); improved cyber-stack performance cited for Terra" + } + ], + "methodology": "Review of the pre-release restricted evaluation period and provider safety statements; the extra scrutiny window is a modest positive signal, but results are not public", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Privacy Policy", + "url": "https://openai.com/policies/privacy-policy", + "date": "2026-07-09", + "value": "No training on API data by default — unchanged for the GPT-5.6 family" + } + ], + "methodology": "Analysis of privacy policies and data handling practices", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 91, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI: Previewing GPT-5.6 Sol", + "url": "https://openai.com/index/previewing-gpt-5-6-sol/", + "date": "2026-06-26", + "value": "Sol explicitly positioned for security-focused applications; staged rollout (partner preview then GA) used for safety evaluation" + } + ], + "methodology": "Safety-stack review plus staged-rollout assessment; full system-card detail for the family was thin on public release day", + "last_verified": "2026-07-09" + }, + "api_security": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Platform Docs", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-07-09", + "value": "API key + OAuth2 authentication, HTTPS only, rate limiting — same hardened platform surface as GPT-5.5" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-07-09" + } + }, + "notes": "The unusual government-requested vetted-partner preview (2026-06-26 to 2026-07-09, ~20 partners, tied to the Cyber EO framework) means the family received extra pre-release scrutiny — but those evaluation results are not public, and independent red-teaming has barely begun. OpenAI has publicly opposed making per-customer government approval permanent." + }, + + "privacy_compliance": { + "overall_score": 87, + "criteria": { + "data_residency": { + "value": "US, EU", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-07-09", + "value": "Data residency options for enterprise customers, unchanged for GPT-5.6" + } + ], + "methodology": "Review of enterprise documentation", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Data Controls", + "url": "https://openai.com/policies/usage-policies", + "date": "2026-07-09", + "value": "API data not used for training by default" + } + ], + "methodology": "Policy review of data usage terms", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "30 days (zero retention available)", + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-07-09", + "value": "30-day default API log retention; zero-data-retention options for qualifying customers" + } + ], + "methodology": "Terms of service and enterprise documentation review", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Tools", + "url": "https://platform.openai.com/docs/guides/safety", + "date": "2026-07-09", + "value": "Customer responsible for PII redaction; moderation API available" + } + ], + "methodology": "Review of data protection capabilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Trust Center", + "url": "https://trust.openai.com/", + "date": "2026-07-09", + "value": "SOC 2 Type II, ISO 27001, GDPR compliant (organization-level; applies to the GPT-5.6 endpoints)" + } + ], + "methodology": "Verification of compliance certifications", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Enterprise", + "url": "https://openai.com/enterprise", + "date": "2026-07-09", + "value": "Zero-data-retention options available for enterprise and qualifying API customers" + } + ], + "methodology": "Enterprise feature review", + "last_verified": "2026-07-09" + } + }, + "notes": "Standard OpenAI enterprise posture, identical to GPT-5.5: SOC 2/ISO 27001, no API-data training by default, 30-day default retention with zero-retention options. Not HIPAA eligible." + }, + + "trust_transparency": { + "overall_score": 87, + "criteria": { + "explainability": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI: Previewing GPT-5.6 Sol", + "url": "https://openai.com/index/previewing-gpt-5-6-sol/", + "date": "2026-06-26", + "value": "Adjustable reasoning effort including a max mode on Sol, plus an Ultra mode that orchestrates subagents with step-level visibility" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 88, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI: Previewing GPT-5.6 Sol", + "url": "https://openai.com/index/previewing-gpt-5-6-sol/", + "date": "2026-06-26", + "value": "Builds on GPT-5.5's factuality improvements; family-specific hallucination data not yet published" + } + ], + "methodology": "Inherited-lineage assessment; no independent factual-QA measurement exists on public release day", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 84, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI Safety", + "url": "https://openai.com/safety", + "date": "2026-06-26", + "value": "Standard bias testing and red-teaming program; GPT-5.6-specific results not yet public" + } + ], + "methodology": "Bias benchmark disclosure review; family-specific data pending", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 88, + "confidence": "low", + "evidence": [ + { + "source": "OpenAI Documentation", + "url": "https://platform.openai.com/docs/models", + "date": "2026-07-09", + "value": "Calibration expected to continue GPT-5.5's trajectory; not yet independently measured for GPT-5.6" + } + ], + "methodology": "Qualitative assessment; launch-day confidence necessarily low", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Help Center: A preview of GPT-5.6 Sol, Terra, and Luna", + "url": "https://help.openai.com/en/articles/20001325-a-preview-of-gpt-56-sol-terra-and-luna", + "date": "2026-07-09", + "value": "Preview post and help-center article cover tiers, pricing, and rollout, but publish fewer benchmark and safety specifics than the GPT-5.5 launch documentation; key specs (context window) unconfirmed" + } + ], + "methodology": "Documentation completeness review against OpenAI's own GPT-5.5 baseline", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 78, + "confidence": "low", + "evidence": [ + { + "source": "ExplainX GPT-5.6 guide", + "url": "https://explainx.ai/blog/gpt-5-6-release-date-features-benchmarks-2026", + "date": "2026-07-09", + "value": "Knowledge cutoff of approximately May 2026 reported but not officially confirmed; training sources not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Safety Systems", + "url": "https://openai.com/safety", + "date": "2026-06-26", + "value": "Multi-layer safety guardrails with agentic-workflow protections; improved cyber-stack behavior cited during the vetted-partner preview" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-07-09" + } + }, + "notes": "Launch documentation is thinner than OpenAI's GPT-5.5 standard: pricing and tiering are clear, but the context window (~1.5M) and knowledge cutoff (~May 2026) remain unconfirmed by official docs, and the government-linked preview evaluations are not public." + }, + + "operational_excellence": { + "overall_score": 90, + "criteria": { + "api_design_quality": { + "score": 95, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI API Reference", + "url": "https://platform.openai.com/docs/api-reference", + "date": "2026-07-09", + "value": "Responses API with streaming, function calling, vision, reasoning-effort control; GPT-5.6 adds Sol max-effort and Ultra subagent modes on the same surface" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI SDKs", + "url": "https://github.com/openai", + "date": "2026-07-09", + "value": "Official SDKs (Python, Node.js, Go, .NET) with day-one GPT-5.6 model-string support" + } + ], + "methodology": "SDK quality, documentation, and maintenance review", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Deprecations", + "url": "https://developers.openai.com/api/docs/deprecations", + "date": "2026-07-09", + "value": "GPT-5.5 remains the designated migration target for the GPT-5.x line; no deprecations triggered by the GPT-5.6 launch, giving adopters a low-pressure upgrade path" + } + ], + "methodology": "Review of versioning policy; the three-tier naming (Sol/Terra/Luna) is new and its long-term versioning behavior is unproven", + "last_verified": "2026-07-09" + }, + "monitoring_observability": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Dashboard", + "url": "https://platform.openai.com/usage", + "date": "2026-07-09", + "value": "Detailed usage dashboard with costs, tokens, and rate limits, covering the new family from day one" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "OpenAI Help Center", + "url": "https://help.openai.com/en/articles/20001325-a-preview-of-gpt-56-sol-terra-and-luna", + "date": "2026-07-09", + "value": "Dedicated help-center guidance for the GPT-5.6 preview and rollout; 24/7 support and active developer community" + } + ], + "methodology": "Support and documentation assessment", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Developer Community announcement", + "url": "https://community.openai.com/t/introducing-gpt-5-6-series-sol-terra-and-luna/1384931", + "date": "2026-07-08", + "value": "Public launch 2026-07-09 after a ~20-partner preview; Cerebras high-speed serving for Sol rolling out through July" + } + ], + "methodology": "Availability-surface analysis; ecosystem score held down because public access began today and third-party integrations are still switching over", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "OpenAI Terms", + "url": "https://openai.com/policies/terms-of-use", + "date": "2026-07-09", + "value": "Standard commercial terms; the government-requested vetted-partner mechanism ended at GA, but OpenAI notes it is resisting any permanent per-customer approval regime" + } + ], + "methodology": "Review of licensing terms; slight residual uncertainty from the Cyber EO framework (2026-08-01 deadline) and potential future access conditions", + "last_verified": "2026-07-09" + } + }, + "notes": "OpenAI's operational machine is mature, but the GPT-5.6 family is hours old publicly: expect fast-moving docs, integration lag, and possible regulatory follow-on from the Cyber EO framework. Sol on Cerebras (up to 750 tok/s) is a July rollout, not universal." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 95, + "notes": "Sol's 88.8% Terminal-Bench 2.1 (91.9% Ultra) edges GPT-5.5, and Luna offers surprising coding value at $1/$6 — but all numbers are provider-reported on launch day.", + "alternatives": ["claude-sonnet-5", "gpt-5-3-codex"] + }, + "customer-support": { + "overall": 92, + "notes": "Terra ($2.50/$15) and Luna ($1/$6) give attractive support tiers if the GPT-5.5-class claim for Terra holds; wait for independent verification before large migrations.", + "alternatives": ["claude-sonnet-5", "gemini-3-5-flash"] + }, + "content-creation": { + "overall": 92, + "notes": "Expected to match or exceed GPT-5.5's strong drafting; family-specific writing evaluations not yet available.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "data-analysis": { + "overall": 93, + "notes": "Sol targets demanding analytical work, and the reported ~1.5M context would be class-leading — but that figure is unconfirmed.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "research-assistant": { + "overall": 93, + "notes": "Promising for literature-scale work if the context claim verifies; GPT-5.5 remains the battle-tested choice this week.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "legal-compliance": { + "overall": 85, + "notes": "Standard OpenAI compliance posture (SOC 2, zero-retention options, not HIPAA eligible); day-one models are a hard sell for conservative legal teams.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-5"] + }, + "healthcare": { + "overall": 81, + "notes": "Not HIPAA eligible, and no clinical validation exists for a model released today.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-5"] + }, + "financial-analysis": { + "overall": 93, + "notes": "Sol's max-effort reasoning targets exactly this workload; quantitative benchmark verification still pending.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "education": { + "overall": 92, + "notes": "Luna's $1/$6 pricing could make high-volume tutoring very economical; content-quality validation pending.", + "alternatives": ["gpt-5-4", "gemini-3-5-flash"] + }, + "creative-writing": { + "overall": 90, + "notes": "No family-specific creative evaluations yet; inherits GPT-5.5's strong narrative baseline with its conciseness bias.", + "alternatives": ["claude-opus-4-8", "gpt-5-5"] + } + }, + + "strengths": [ + "Three clean price tiers: Sol $5/$30 (flagship), Terra $2.50/$15, Luna $1/$6 per 1M tokens", + "Sol beats GPT-5.5 on Terminal-Bench 2.1 (88.8% vs 88.0%; 91.9% in Ultra subagent mode)", + "Terra reported to deliver GPT-5.5-class capability at half the price", + "Reported ~1.5M token context would be class-leading (unconfirmed)", + "Extra pre-release scrutiny via the US-government-requested vetted-partner preview", + "Sol on Cerebras at up to 750 tokens/s rolling out through July 2026", + "Same mature Responses API/SDK surface as GPT-5.5 — trivial migration" + ], + + "limitations": [ + "Publicly released today (2026-07-09) — essentially no independent benchmarks, latency data, or production track record", + "Headline ~1.5M context and ~May 2026 knowledge cutoff are reported but not officially confirmed", + "Terra-equals-GPT-5.5-at-half-price claim is provider/partner-sourced and unverified", + "Reported tier ordering is inconsistent (Luna 84.3% vs Terra 82.5% on Terminal-Bench 2.1) — treat tier choices as provisional", + "Regulatory overhang: launched under the US Cyber EO framework (2026-08-01 deadline); access conditions could evolve", + "Not HIPAA eligible; 30-day default API retention", + "Launch documentation thinner than OpenAI's GPT-5.5 standard" + ], + + "best_for": [ + "Early adopters wanting frontier agentic coding with a cheap fallback path to GPT-5.5", + "Tiered deployments matching workload difficulty to Sol/Terra/Luna price points", + "Extended agent-driven workflows using Sol's max-effort and Ultra subagent modes", + "High-volume, cost-sensitive inference on Luna ($1/$6) pending quality verification", + "Teams already on the Responses API who can A/B against GPT-5.5 immediately" + ], + + "not_recommended_for": [ + "Production-critical workloads until independent benchmarks and a few weeks of track record exist", + "HIPAA-regulated healthcare applications", + "Architectures that depend on the unconfirmed ~1.5M context window", + "Conservative regulated-industry deployments given the unresolved Cyber EO access framework" + ], + + "metadata": { + "pricing": { + "input": "$5.00 per 1M tokens (Sol)", + "output": "$30.00 per 1M tokens (Sol)", + "notes": "Family pricing per 1M tokens: Sol $5/$30, Terra $2.50/$15, Luna $1/$6 (confirmed across launch coverage 2026-07-09). Sol matches GPT-5.5's price point. Batch/caching discounts expected to mirror GPT-5.5 but not yet verified for the new family.", + "last_verified": "2026-07-09" + }, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Russian", + "Arabic", + "Hindi", + "50+ languages" + ], + "modalities": ["text", "vision"], + "api_endpoint": "https://api.openai.com/v1/responses", + "open_source": false, + "architecture": "Transformer-based three-tier family (Sol flagship / Terra balanced / Luna fast) with adjustable reasoning effort, Sol max-effort mode, and Ultra subagent mode", + "parameters": "Not disclosed", + "knowledge_cutoff": "~May 2026 (reported, not officially confirmed)", + "release_date": "2026-07-09 (public; partner-only preview from 2026-06-26)", + "variants": "gpt-5.6-sol (flagship), gpt-5.6-terra (balanced), gpt-5.6-luna (fast/affordable)" + }, + + "related_entities": ["gpt-5-5", "gpt-5-4", "gpt-5-3-codex", "claude-sonnet-5", "gemini-3-1-pro"], + + "tags": [ + "flagship", + "variant-family", + "agentic", + "reasoning", + "launch-day", + "government-preview", + "tiered-pricing", + "token-efficient" + ] +} diff --git a/data/models/grok-4-5.json b/data/models/grok-4-5.json new file mode 100644 index 0000000..da70440 --- /dev/null +++ b/data/models/grok-4-5.json @@ -0,0 +1,661 @@ +{ + "id": "grok-4-5", + "type": "model", + "name": "Grok 4.5", + "provider": "SpaceXAI (formerly xAI)", + "version": "4.5", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "SpaceXAI's flagship (released 2026-07-08, days after the xAI-to-SpaceXAI rebrand): an 'Opus-class' model tuned for token efficiency at $2/$6 per 1M ($0.50 cached), 500K context, reasoning effort low/medium/high. Strong day-one results (83.3% Terminal-Bench 2.1, #4 on Artificial Analysis Intelligence Index) but no EU availability at launch, thin enterprise compliance, and a provider under active regulatory investigation over Grok content-safety failures.", + "website": "https://docs.x.ai/developers/models/grok-4.5", + "trust_vector": { + "performance_reliability": { + "overall_score": 93, + "criteria": { + "task_accuracy_code": { + "score": 95, + "confidence": "medium", + "evidence": [ + { + "source": "Kingy.ai Grok 4.5 benchmark analysis", + "url": "https://kingy.ai/blog/grok-4-5-benchmarks-pricing-context-window/", + "date": "2026-07-08", + "value": "83.3% Terminal-Bench 2.1 (vs Opus 4.8's 78.9%, Fable 5's 84.3%); 64.7% SWE-Bench Pro (vs Opus 4.8's 69.2%); DeepSWE results split by harness (62.0% provider harness vs 53% neutral harness)" + }, + { + "source": "xAI/SpaceXAI Release Notes", + "url": "https://docs.x.ai/developers/release-notes", + "date": "2026-07-08", + "value": "Positioned as SpaceXAI's smartest model for coding, agentic tasks, and knowledge work" + } + ], + "methodology": "Provider benchmarks cross-checked against third-party analysis; harness-dependent DeepSWE split means no universal supremacy claim holds", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis (via Kingy.ai)", + "url": "https://kingy.ai/blog/grok-4-5-benchmarks-pricing-context-window/", + "date": "2026-07-08", + "value": "Artificial Analysis Intelligence Index 54, ranked #4 overall behind Fable 5 (60), Opus 4.8 (56), and GPT-5.5 (55)" + } + ], + "methodology": "Independent aggregator index placement on launch day; Musk's internal characterization ('roughly comparable to Opus 4.7, but much faster') treated as marketing", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "Axios", + "url": "https://www.axios.com/2026/07/08/spacexai-grok-new-model", + "date": "2026-07-08", + "value": "Launched 2026-07-08 after a private beta with SpaceX and Tesla teams; billed as an Opus-class model that is faster, more token-efficient, and lower cost" + }, + { + "source": "OpenRouter Model Listing", + "url": "https://openrouter.ai/x-ai/grok-4.5", + "date": "2026-07-09", + "value": "Listed day-one with competitive quality ratings; Coding Agent Index 76 per Artificial Analysis" + } + ], + "methodology": "Launch coverage and aggregator listings; broad task coverage not yet independently characterized", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 91, + "confidence": "medium", + "evidence": [ + { + "source": "Kingy.ai Grok 4.5 benchmark analysis", + "url": "https://kingy.ai/blog/grok-4-5-benchmarks-pricing-context-window/", + "date": "2026-07-08", + "value": "Exceptional token efficiency: ~15,954 average output tokens per SWE-Bench Pro task vs 67,020 for Opus 4.8 max (~4.2x); structured outputs and function calling supported" + } + ], + "methodology": "Review of structured-output features and provider token-efficiency reports; repeated-run data pending", + "last_verified": "2026-07-09" + }, + "latency_p50": { + "value": "~80-91 output tokens/s; full-response times competitive at default high effort", + "confidence": "low", + "evidence": [ + { + "source": "Kingy.ai Grok 4.5 benchmark analysis", + "url": "https://kingy.ai/blog/grok-4-5-benchmarks-pricing-context-window/", + "date": "2026-07-08", + "value": "Official output speed 80 tokens/s; independently measured 91.3 tokens/s on launch day" + } + ], + "methodology": "Day-one throughput measurements; percentile latency distributions not yet available", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "Unknown — reasoning cannot be disabled, so tail latency will track effort setting", + "confidence": "low", + "evidence": [ + { + "source": "xAI/SpaceXAI Release Notes", + "url": "https://docs.x.ai/developers/release-notes", + "date": "2026-07-08", + "value": "Configurable reasoning effort (low/medium/high, default high); no p95 data exists one day post-launch" + } + ], + "methodology": "95th percentile latency requires post-launch measurement", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "500,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "SpaceXAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.5", + "date": "2026-07-09", + "value": "500K token context window (down from Grok 4.3's 1M); higher-context rates apply above 200K-token requests" + } + ], + "methodology": "Official specification from provider documentation", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 92, + "confidence": "low", + "evidence": [ + { + "source": "xAI Status Page", + "url": "https://status.x.ai/", + "date": "2026-07-09", + "value": "Platform generally stable; Grok 4.5 endpoints (us-east-1, us-west-2) live since 2026-07-08 with no model-specific track record, and no EU serving at launch" + } + ], + "methodology": "Platform status baseline; model is one day old and rate-limited at 150 req/s and 50M tokens/min", + "last_verified": "2026-07-09" + } + }, + "notes": "Genuinely strong day-one showing — beats Opus 4.8 on Terminal-Bench 2.1 and delivers ~4.2x better token efficiency — but it is not the benchmark leader (Fable 5 and Opus 4.8 rank above it on the Artificial Analysis Intelligence Index), and the DeepSWE provider-vs-neutral harness split (62.0% vs 53%) warrants caution about provider-reported numbers. Context window halved vs Grok 4.3 (500K vs 1M)." + }, + "security": { + "overall_score": 82, + "criteria": { + "prompt_injection_resistance": { + "score": 84, + "confidence": "low", + "evidence": [ + { + "source": "SpaceXAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.5", + "date": "2026-07-09", + "value": "Hardened system-prompt handling carried over from the Grok 4.x line; no Grok 4.5-specific red-team data published" + } + ], + "methodology": "Inherited-lineage review against OWASP LLM01 patterns; provider publishes little red-team detail", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 84, + "confidence": "low", + "evidence": [ + { + "source": "xAI/SpaceXAI News", + "url": "https://x.ai/news", + "date": "2026-07-08", + "value": "Safety improvements cited at launch; substantially less published adversarial testing than Anthropic/OpenAI/Google" + } + ], + "methodology": "Review of provider claims and community jailbreak reports; day-one data is minimal", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2026-07-08", + "value": "API data handling documented; fewer contractual controls than major enterprise providers" + } + ], + "methodology": "Analysis of privacy policies and data handling commitments", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 77, + "confidence": "medium", + "evidence": [ + { + "source": "TechPolicy.Press — Regulators Are Going After Grok and X", + "url": "https://www.techpolicy.press/regulators-are-going-after-grok-and-x-just-not-together/", + "date": "2026-07-09", + "value": "Ofcom and the European Commission have open formal investigations, and Brazil issued a 30-day ultimatum, over Grok's mass generation of sexualized imagery including apparent minors (CCDH: 3M+ sexualized images in under two weeks)" + }, + { + "source": "SpaceXAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.5", + "date": "2026-07-09", + "value": "Content moderation in place for the text API; SpaceXAI publishes far less safety-evaluation detail than peers" + } + ], + "methodology": "Safety review inheriting the active Grok content-safety crisis context (see grok-4-3): the failures center on Grok consumer image products rather than this text API, but Grok 4.5 launched while those investigations remain open and evidences the same provider-level output-safety governance", + "last_verified": "2026-07-09", + "notes": "Scored one point below grok-4-3's post-crisis 78: same provider governance concerns, plus zero model-specific safety documentation for a day-old flagship" + }, + "api_security": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "SpaceXAI API Documentation", + "url": "https://docs.x.ai/", + "date": "2026-07-09", + "value": "API key authentication, HTTPS only, rate limiting (150 req/s, 50M tokens/min), team management in console" + } + ], + "methodology": "Review of API security features and authentication mechanisms", + "last_verified": "2026-07-09" + } + }, + "notes": "Baseline security is reasonable, but Grok 4.5 launched into the middle of the provider's content-safety crisis (Ofcom and European Commission investigations open, Brazil ultimatum) with no model-specific safety card. The regulatory findings concern Grok consumer image products, not this text API, yet they weigh on confidence in SpaceXAI's safety governance." + }, + "privacy_compliance": { + "overall_score": 75, + "criteria": { + "data_residency": { + "value": "US only at launch (us-east-1, us-west-2); no EU availability", + "confidence": "high", + "evidence": [ + { + "source": "SpaceXAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.5", + "date": "2026-07-09", + "value": "Deployed in us-east-1 and us-west-2 only" + }, + { + "source": "Kingy.ai Grok 4.5 benchmark analysis", + "url": "https://kingy.ai/blog/grok-4-5-benchmarks-pricing-context-window/", + "date": "2026-07-08", + "value": "Not available in the EU at launch; EU availability expected mid-July per launch coverage" + } + ], + "methodology": "Review of provider documentation and launch coverage; EU absence is notable given the provider's open European Commission investigation", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2026-07-08", + "value": "API customer data not used for training by default per policy" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "30 days (standard API)", + "confidence": "medium", + "evidence": [ + { + "source": "xAI Privacy Policy", + "url": "https://x.ai/legal/privacy-policy", + "date": "2026-07-08", + "value": "Limited retention for abuse monitoring; zero-retention requires enterprise agreement" + } + ], + "methodology": "Review of terms of service and data retention policies", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "SpaceXAI Documentation", + "url": "https://docs.x.ai/", + "date": "2026-07-09", + "value": "Customer responsible for PII redaction; no built-in PII tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Trust Center", + "url": "https://trust.x.ai/", + "date": "2026-07-09", + "value": "SOC 2 Type II; thinner certification portfolio (no HIPAA BAA program, limited GDPR tooling) vs Anthropic/OpenAI/Google — unchanged through the SpaceXAI rebrand" + } + ], + "methodology": "Verification of compliance certifications against major enterprise provider baselines", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Trust Center", + "url": "https://trust.x.ai/", + "date": "2026-07-09", + "value": "Zero-data-retention available only via negotiated enterprise terms" + } + ], + "methodology": "Review of data handling practices and enterprise contract options", + "last_verified": "2026-07-09" + } + }, + "notes": "Scored one point below grok-4-3 (76 to 75): the compliance portfolio is unchanged, but Grok 4.5 additionally launched US-only with no EU availability — a material constraint for GDPR-scoped workloads, and conspicuous while the European Commission's Grok investigation is open. Corporate-entity churn (xAI absorbed into SpaceX, rebranded SpaceXAI) adds contractual diligence overhead." + }, + "trust_transparency": { + "overall_score": 80, + "criteria": { + "explainability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "xAI/SpaceXAI Release Notes", + "url": "https://docs.x.ai/developers/release-notes", + "date": "2026-07-08", + "value": "Configurable reasoning effort (low/medium/high, default high) with inspectable reasoning" + } + ], + "methodology": "Evaluation of reasoning transparency and explanation capabilities", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 84, + "confidence": "low", + "evidence": [ + { + "source": "xAI/SpaceXAI News", + "url": "https://x.ai/news", + "date": "2026-07-08", + "value": "Continued quality improvements claimed at launch; no independent factual-QA measurement exists yet" + } + ], + "methodology": "Review of provider claims; day-one confidence necessarily low", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "xAI Public Statements", + "url": "https://x.ai/news", + "date": "2026-07-08", + "value": "Limited published bias evaluation; past Grok versions drew scrutiny over politically tuned behavior" + } + ], + "methodology": "Review of bias benchmark disclosures and independent reporting", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 82, + "confidence": "low", + "evidence": [ + { + "source": "SpaceXAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.5", + "date": "2026-07-09", + "value": "Reasoning mode expresses uncertainty reasonably per Grok 4.x lineage; 4.5-specific calibration not yet assessed" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "SpaceXAI Model Documentation", + "url": "https://docs.x.ai/developers/models/grok-4.5", + "date": "2026-07-09", + "value": "Solid developer model page (pricing, limits, features, aliases grok-4.5 / grok-4.5-latest / grok-build-latest), but no safety card; knowledge cutoff and max output not published" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "Kingy.ai Grok 4.5 benchmark analysis", + "url": "https://kingy.ai/blog/grok-4-5-benchmarks-pricing-context-window/", + "date": "2026-07-08", + "value": "Trained on tens of thousands of NVIDIA GB300 GPUs with RL over hundreds of thousands of tasks; Musk self-reported ~1.5T parameters (V9) but SpaceXAI has not confirmed — sources undisclosed" + } + ], + "methodology": "Review of public disclosures; key claims come from executive posts rather than documentation", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 77, + "confidence": "medium", + "evidence": [ + { + "source": "SpaceXAI Documentation", + "url": "https://docs.x.ai/", + "date": "2026-07-09", + "value": "Built-in moderation with developer controls; lighter-touch defaults than peers" + }, + { + "source": "The Conversation — Grok sexualized images AI reckoning", + "url": "https://theconversation.com/the-furore-over-groks-sexualised-images-has-begun-an-ai-reckoning-275448", + "date": "2026-07-01", + "value": "The 2026 Grok controversy showed xAI's guardrails failed at scale on sexualized imagery, including of minors, prompting UK, EU, and Brazilian regulatory action" + } + ], + "methodology": "Analysis of built-in safety mechanisms, inheriting the provider-level guardrail-failure context applied to grok-4-3; no Grok 4.5-specific safety evaluation has been published to offset it", + "last_verified": "2026-07-09" + } + }, + "notes": "Good developer-facing documentation, but transparency gaps are wider than peers at this tier: no safety card, no knowledge cutoff, and headline training claims (1.5T parameters) sourced from executive posts. Inherits the guardrails downgrade applied to grok-4-3 after the Grok content-safety crisis." + }, + "operational_excellence": { + "overall_score": 82, + "criteria": { + "api_design_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "SpaceXAI API Documentation", + "url": "https://docs.x.ai/", + "date": "2026-07-09", + "value": "OpenAI-compatible API with reasoning effort control, function calling, structured outputs, and prompt caching ($0.50 cached input)" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI SDKs", + "url": "https://github.com/xai-org", + "date": "2026-07-09", + "value": "Official SDKs plus broad compatibility with OpenAI client libraries; day-one grok-4.5 support" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Migration Guide (May 15 Retirement)", + "url": "https://docs.x.ai/developers/migration/may-15-retirement", + "date": "2026-07-09", + "value": "Provider history of aggressive retirements with silent slug redirects (retired slugs redirect to grok-4.3); Grok 4.5 supersedes the flagship barely two months after Grok 4.3 shipped, and halves the context window (1M to 500K)" + } + ], + "methodology": "Review of deprecation/migration practices; fast flagship churn plus spec regressions (context) reduce planning predictability", + "last_verified": "2026-07-09", + "notes": "grok-4.5-latest and grok-build-latest aliases repeat the mutable-alias pattern that made prior retirements silent" + }, + "monitoring_observability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "SpaceXAI Console", + "url": "https://console.x.ai/", + "date": "2026-07-09", + "value": "Usage dashboard with spend and rate-limit visibility (150 req/s, 50M tokens/min for grok-4.5)" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "SpaceXAI Documentation", + "url": "https://docs.x.ai/", + "date": "2026-07-09", + "value": "Improving documentation, but support channels remain lighter than major cloud providers, and the corporate transition (xAI to SpaceXAI, completed with the 2026-07-06/07 rebrand) is still settling" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness amid the merger transition", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 83, + "confidence": "medium", + "evidence": [ + { + "source": "Kingy.ai Grok 4.5 benchmark analysis", + "url": "https://kingy.ai/blog/grok-4-5-benchmarks-pricing-context-window/", + "date": "2026-07-08", + "value": "Day-one availability across Grok Build, Cursor, the xAI API console, Microsoft Office add-ins, OpenRouter, Vercel, Cloudflare, Snowflake, and Databricks Mosaic — but not in the EU" + } + ], + "methodology": "Analysis of third-party integrations; unusually broad launch surface, offset by the EU gap", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "xAI Terms of Service", + "url": "https://x.ai/legal/terms-of-service", + "date": "2026-07-09", + "value": "Clear commercial API terms; enterprise agreements available. Contracting entity is transitioning to SpaceXAI following the merger (rebrand announced 2026-07-06, completed 2026-07-07)" + } + ], + "methodology": "Review of licensing terms and restrictions; entity-change diligence recommended for existing contracts", + "last_verified": "2026-07-09" + } + }, + "notes": "Impressively broad day-one distribution (Cursor, OpenRouter, Vercel, Cloudflare, Snowflake, Databricks, Office add-ins) and a clean OpenAI-compatible API. Counterweights: no EU serving at launch, flagship churn (two flagships in ~10 weeks), a halved context window vs Grok 4.3, and a mid-merger provider — xAI is now SpaceXAI under SpaceX after the February 2026 acquisition, with the rebrand completed 2026-07-07. API endpoints and docs remain on x.ai domains." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "Beats Opus 4.8 on Terminal-Bench 2.1 (83.3% vs 78.9%) with ~4.2x better token efficiency at a fraction of the price; Fable 5 and Opus 4.8 still lead on the hardest work, and neutral-harness results are less flattering.", + "alternatives": ["claude-sonnet-5", "claude-opus-4-8"] + }, + "customer-support": { + "overall": 86, + "notes": "Cheap, fast, and token-efficient for support; compliance posture and no EU serving limit regulated or European deployments.", + "alternatives": ["claude-sonnet-5", "gemini-3-5-flash"] + }, + "content-creation": { + "overall": 88, + "notes": "Strong generation with current-events awareness from the X ecosystem; brand-safety review advisable given the provider's ongoing content-safety controversy.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + }, + "data-analysis": { + "overall": 89, + "notes": "Strong reasoning per dollar ($0.31 per Artificial Analysis Intelligence Index task); 500K context is half of Grok 4.3's 1M, with higher rates above 200K.", + "alternatives": ["gemini-3-1-pro", "grok-4-3"] + }, + "research-assistant": { + "overall": 89, + "notes": "Excellent cost-efficiency for synthesis; for literature-scale corpora above 500K tokens, Grok 4.3's 1M window or Gemini remain better fits.", + "alternatives": ["gemini-3-1-pro", "grok-4-3"] + }, + "legal-compliance": { + "overall": 74, + "notes": "Thin certifications, no EU availability, and an actively investigated provider make this a poor fit for compliance-sensitive legal work.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-5"] + }, + "healthcare": { + "overall": 70, + "notes": "No HIPAA eligibility program; not recommended for PHI workloads.", + "alternatives": ["claude-opus-4-8", "claude-sonnet-5"] + }, + "financial-analysis": { + "overall": 88, + "notes": "Strong quantitative reasoning and token efficiency at $2/$6; verify compliance requirements and day-one stability first.", + "alternatives": ["gpt-5-5", "gemini-3-1-pro"] + }, + "education": { + "overall": 88, + "notes": "Strong explanations at low cost; content controls are lighter-touch than peers, which matters for minor-facing deployments given the provider's 2026 safety record.", + "alternatives": ["claude-sonnet-5", "gemini-3-5-flash"] + }, + "creative-writing": { + "overall": 88, + "notes": "Distinctive voice and fewer content restrictions than competitors; day-one creative evaluations are anecdotal.", + "alternatives": ["gpt-5-5", "claude-opus-4-8"] + } + }, + "strengths": [ + "Standout intelligence-per-dollar: $2/$6 per 1M with $0.50 cached input and ~$0.31 per Artificial Analysis Intelligence Index task", + "Exceptional token efficiency: ~15,954 output tokens per SWE-Bench Pro task vs 67,020 for Opus 4.8 max (~4.2x)", + "Beats Opus 4.8 on Terminal-Bench 2.1 (83.3% vs 78.9%); #4 on the Artificial Analysis Intelligence Index at launch", + "Full agentic feature set: reasoning effort control, function calling, structured outputs, prompt caching", + "Unusually broad day-one distribution: Cursor, OpenRouter, Vercel, Cloudflare, Snowflake, Databricks, Office add-ins", + "OpenAI-compatible API simplifies migration" + ], + "limitations": [ + "No EU availability at launch (US-only serving in us-east-1/us-west-2; EU expected mid-July per launch coverage)", + "Provider under active regulatory investigation (Ofcom, European Commission; Brazil ultimatum) over Grok content-safety failures; no Grok 4.5 safety card published", + "Context window halved vs Grok 4.3 (500K vs 1M), with higher per-token rates above 200K", + "Released 2026-07-08 — benchmarks are largely provider-reported, and the DeepSWE provider-vs-neutral harness gap (62.0% vs 53%) urges caution", + "Thin enterprise compliance: SOC 2 only, no HIPAA program, zero-retention only via negotiated terms", + "Provider turbulence: xAI merged into SpaceX and completed the SpaceXAI rebrand 2026-07-07, one day before launch; reasoning cannot be fully disabled", + "Flagship churn: second flagship in ~10 weeks, with a history of silent slug redirects on retirement" + ], + "best_for": [ + "Cost-sensitive agentic coding needing near-frontier quality with minimal token spend", + "High-volume agent fleets where the ~4.2x token-efficiency advantage compounds", + "US-based teams on OpenAI-compatible tooling seeking cheaper frontier capacity", + "Applications wanting real-time/current-events awareness from the X ecosystem" + ], + "not_recommended_for": [ + "EU-based or GDPR-scoped workloads (no EU serving at launch)", + "Healthcare workloads involving PHI (no HIPAA eligibility)", + "Regulated industries requiring deep compliance attestations or a stable contracting entity", + "Workloads needing more than 500K context (use Grok 4.3 or Gemini)", + "Risk-averse teams unwilling to bet on a day-old model from a provider under active safety investigations" + ], + "metadata": { + "pricing": { + "input": "$2.00 per 1M tokens", + "output": "$6.00 per 1M tokens", + "notes": "Cached input $0.50 per 1M tokens. Higher per-token rates apply for requests above 200K context. Verified against docs.x.ai model page 2026-07-09. Roughly 1.6x Grok 4.3's input price but with much stronger per-task economics via token efficiency.", + "last_verified": "2026-07-09" + }, + "context_window": 500000, + "languages": [ + "English", + "Spanish", + "French", + "German", + "Italian", + "Portuguese", + "Japanese", + "Korean", + "Chinese", + "Arabic", + "Hindi" + ], + "modalities": ["text", "image (input)"], + "api_endpoint": "https://api.x.ai/v1/chat/completions", + "api_model_id": "grok-4.5 (aliases: grok-4.5-latest, grok-build-latest)", + "open_source": false, + "architecture": "Transformer-based with always-on configurable reasoning (low/medium/high, default high), function calling, and structured outputs; Musk self-reported ~1.5T parameters (unconfirmed)", + "parameters": "~1.5T claimed by Musk; not officially confirmed", + "release_date": "2026-07-08", + "lifecycle_status": "Current SpaceXAI flagship; Grok 4.3 remains served below it. US-only serving at launch (us-east-1, us-west-2)." + }, + "related_entities": ["grok-4-3", "grok-4-1", "claude-sonnet-5", "gpt-5-6", "claude-opus-4-8"], + "tags": [ + "reasoning", + "token-efficient", + "cost-effective", + "function-calling", + "structured-outputs", + "real-time", + "new-release", + "regulatory-scrutiny", + "no-eu-availability" + ] +} diff --git a/data/models/kimi-k2-7-code.json b/data/models/kimi-k2-7-code.json new file mode 100644 index 0000000..6596530 --- /dev/null +++ b/data/models/kimi-k2-7-code.json @@ -0,0 +1,640 @@ +{ + "id": "kimi-k2-7-code", + "type": "model", + "name": "Kimi K2.7-Code", + "provider": "Moonshot AI", + "version": "20260612", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Moonshot AI's coding-specialized open-weight MoE (1T total / 32B active, Modified MIT) built on Kimi K2.6, released 2026-06-12. Vendor reports 62.0 on Kimi Code Bench v2 (+21.8% over K2.6) and ~30% lower reasoning-token usage, but all published benchmarks are Moonshot-run with no independent public-suite results yet. 256K context, thinking mode always on, $0.95/$4.00 per 1M tokens.", + "website": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + + "trust_vector": { + "performance_reliability": { + "overall_score": 90, + "criteria": { + "task_accuracy_code": { + "score": 94, + "confidence": "low", + "evidence": [ + { + "source": "MarkTechPost release coverage (vendor-reported)", + "url": "https://www.marktechpost.com/2026/06/12/moonshot-ai-releases-kimi-k2-7-code-a-coding-model-reporting-21-8-on-kimi-code-bench-v2-over-k2-6/", + "date": "2026-06-12", + "value": "Kimi Code Bench v2: 62.0 vs K2.6's 50.9 (+21.8%); MCP Mark Verified 81.1 vs Claude Opus 4.8's 76.4; trails GPT-5.5 on most vendor-run metrics" + }, + { + "source": "DevOps.com analysis", + "url": "https://devops.com/moonshot-ais-kimi-k2-7-code-targets-token-efficiency-in-agentic-coding/", + "date": "2026-06-13", + "value": "Positioning is token efficiency in agentic coding rather than raw benchmark leadership; all published benchmarks are vendor-run, not independent" + } + ], + "methodology": "Vendor-run proprietary benchmarks only (Kimi Code Bench v2, MCP Mark Verified); no public-suite (SWE-bench) results or independent replication yet — scored by anchoring to K2.6's verified base with the specialization claim discounted", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 88, + "confidence": "low", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Built on K2.6's base with mandatory thinking mode; coding specialization trades some general reasoning breadth; ~30% lower reasoning-token usage claimed at comparable quality" + } + ], + "methodology": "Inference from K2.6 lineage and vendor claims; no independent reasoning benchmarks published for K2.7-Code", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 84, + "confidence": "low", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Coding-focused specialization of K2.6; general-domain performance not a design goal and not benchmarked by the vendor" + } + ], + "methodology": "Review of vendor positioning; general capability anchored slightly below the K2.6 generalist base", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 84, + "confidence": "low", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Sampling parameters fixed server-side and thinking mode mandatory, which promotes run-to-run consistency; only weeks of community testing so far" + } + ], + "methodology": "Early community testing of repeated runs and agent trajectories; limited observation window since June launch", + "last_verified": "2026-07-09" + }, + "latency_p50": { + "value": "2.8s", + "confidence": "low", + "evidence": [ + { + "source": "OpenRouter model page", + "url": "https://openrouter.ai/moonshotai/kimi-k2.7-code", + "date": "2026-06-20", + "value": "Comparable per-request latency to K2.6 (32B active); ~30% fewer reasoning tokens shortens end-to-end agentic turns despite mandatory thinking mode" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes; varies widely by host", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "7.0s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://openrouter.ai/moonshotai/kimi-k2.7-code", + "date": "2026-06-20", + "value": "p95 ~7.0s; long agentic coding chains take substantially longer by design" + } + ], + "methodology": "95th percentile response time from early third-party measurements", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "262,144 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "256K (262,144) token context window, unchanged from K2.6" + } + ], + "methodology": "Official specification from model card, confirmed by OpenRouter listing (262K)", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 93, + "confidence": "low", + "evidence": [ + { + "source": "Moonshot AI Platform", + "url": "https://platform.moonshot.ai/", + "date": "2026-07-01", + "value": "Served on the same first-party platform as K2.6 (generally stable); open weights allow self-hosted redundancy; no long-run record for this endpoint yet" + } + ], + "methodology": "Review of platform availability and self-hosting fallback options; observation window under one month", + "last_verified": "2026-07-09" + } + }, + "notes": "Vendor claims meaningful coding gains over the already-strong K2.6 (62.0 vs 50.9 on Kimi Code Bench v2) plus ~30% lower reasoning-token usage, but every published number is Moonshot-run on proprietary benchmarks — no SWE-bench or other public-suite results exist yet. Scores anchor to K2.6's verified base with the uplift discounted until independent replication." + }, + + "security": { + "overall_score": 78, + "criteria": { + "prompt_injection_resistance": { + "score": 77, + "confidence": "low", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Inherits K2.6 safety tuning; no published third-party prompt-injection audit; agentic coding use (tool execution, repo access) raises injection stakes" + } + ], + "methodology": "Review of vendor safety documentation and K2.6 precedent against OWASP LLM01 patterns; model too new for mature red-team coverage", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "Community red-teaming", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-25", + "value": "Standard alignment tuning; open weights mean guardrails can be removed in fine-tuned derivatives; limited adversarial testing published since launch" + } + ], + "methodology": "Early testing against adversarial prompt datasets; open-weight deployments inherit deployer responsibility", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Privacy Policy", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-06-12", + "value": "Standard data handling on first-party API (same platform and policies as K2.6); full control when self-hosted" + } + ], + "methodology": "Analysis of privacy policies and self-hosting data-control options", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Safety post-training inherited from K2.6; refusal behavior comparable to other open frontier models per early reports" + } + ], + "methodology": "Safety testing across harmful content categories per vendor card and K2.6 precedent", + "last_verified": "2026-07-09" + }, + "api_security": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI API Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-06-12", + "value": "API key authentication, HTTPS only, rate limiting; OpenAI-compatible endpoints shared with the K2 family" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-07-09" + } + }, + "notes": "Inherits K2.6's standard open-model posture with no third-party audit. As a coding agent typically wired to tool execution and repositories, deployers should treat prompt-injection hardening as their own responsibility." + }, + + "privacy_compliance": { + "overall_score": 74, + "criteria": { + "data_residency": { + "value": "China (first-party API); any jurisdiction when self-hosted or via Western hosts", + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Platform Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-06-12", + "value": "Moonshot AI is a China-based provider; first-party API data processed under Chinese jurisdiction" + }, + { + "source": "OpenRouter availability", + "url": "https://openrouter.ai/moonshotai/kimi-k2.7-code", + "date": "2026-06-20", + "value": "Available via OpenRouter and Western inference hosts within days of release, enabling non-China residency" + } + ], + "methodology": "Review of provider jurisdiction and third-party hosting options", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Privacy Policy", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-06-12", + "value": "API data usage terms standard for the segment (unchanged from K2.6); self-hosting removes the question entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "Per Moonshot policy on first-party API (China jurisdiction); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Terms", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-06-12", + "value": "First-party retention governed by Chinese data regulations; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service and deployment-dependent retention", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 72, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-06-12", + "value": "Customer responsible for PII redaction; no managed PII tooling; source code sent to the API may itself contain secrets and PII" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 68, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI public materials", + "url": "https://www.moonshot.ai/", + "date": "2026-06-12", + "value": "No published SOC 2 / HIPAA / GDPR attestations for the first-party API; Western hosts may carry their own certifications" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Open weights on Hugging Face", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Self-hosting via vLLM/SGLang (~595 GB on disk) gives complete data control and zero external retention — attractive for keeping proprietary code in-house" + } + ], + "methodology": "Review of self-hosting deployment options enabling zero retention", + "last_verified": "2026-07-09" + } + }, + "notes": "Same posture as K2.6: first-party API under Chinese jurisdiction is a material caveat — amplified for this model because coding workloads routinely transmit proprietary source code. Open weights fully mitigate for organizations able to self-host or use Western hosts." + }, + + "trust_transparency": { + "overall_score": 77, + "criteria": { + "explainability": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Thinking mode is mandatory, so reasoning traces are always available; tool-call trajectories inspectable as with K2.6's agentic stack" + } + ], + "methodology": "Evaluation of reasoning and agent-trajectory transparency", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 79, + "confidence": "low", + "evidence": [ + { + "source": "Community testing", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-25", + "value": "Early reports mirror K2.6: moderate hallucination, with tool-use grounding improving factuality in agentic coding loops" + } + ], + "methodology": "Early testing on factual QA and tool-augmented coding workflows", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Limited published bias evaluation; lower salience for a coding-specialized model but unevaluated nonetheless" + } + ], + "methodology": "Review of published bias benchmarks and community evaluations", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 77, + "confidence": "low", + "evidence": [ + { + "source": "Model behavior testing", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-25", + "value": "Expresses uncertainty adequately in thinking traces; no calibrated confidence outputs" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face model card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Detailed card: 1T/32B MoE, 61 layers, 384 experts, MLA attention, 400M MoonViT vision encoder, deployment guides; benchmark section limited to Moonshot-proprietary suites" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI publications", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Architecture well documented; the coding-specialization training recipe and data composition on top of K2.6 are not disclosed" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 77, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi K2.7-Code Model Card", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Built-in safety tuning inherited from K2.6; deployers of open weights must layer their own guardrails, especially around code-execution tools" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-07-09" + } + }, + "notes": "Open weights, a detailed architecture card, and always-on thinking traces give decent transparency, but the evaluation story is weaker than K2.6's: all benchmarks are Moonshot-proprietary (Kimi Code Bench v2, MCP Mark Verified) with no public-suite or independent results yet." + }, + + "operational_excellence": { + "overall_score": 81, + "criteria": { + "api_design_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "Moonshot AI API Documentation", + "url": "https://platform.moonshot.ai/docs", + "date": "2026-06-12", + "value": "OpenAI-compatible API with streaming, tool calling, vision input, and prompt caching; constraint: thinking mode is mandatory and sampling parameters are fixed server-side" + } + ], + "methodology": "Review of API design, consistency, and feature completeness; server-side sampling constraints reduce configurability", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI GitHub / Kimi Code", + "url": "https://github.com/MoonshotAI", + "date": "2026-06-12", + "value": "OpenAI-compatible so mainstream SDKs work; first-party Kimi Code CLI ships alongside the model for agentic coding workflows" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Kimi release history", + "url": "https://huggingface.co/moonshotai", + "date": "2026-06-12", + "value": "K2.7-Code is a coding-specialized sibling built on K2.6 (released eight weeks prior), not a replacement — K2.6 remains the general-purpose flagship and prior weights stay available; cadence is fast" + } + ], + "methodology": "Review of versioning practices and weight availability", + "last_verified": "2026-07-09" + }, + "monitoring_observability": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI Platform", + "url": "https://platform.moonshot.ai/", + "date": "2026-06-12", + "value": "Basic usage dashboard; self-hosted observability is deployer-built" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Moonshot AI community channels", + "url": "https://github.com/MoonshotAI", + "date": "2026-06-12", + "value": "GitHub and community support; limited English-language enterprise support" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenRouter and inference ecosystem", + "url": "https://openrouter.ai/moonshotai/kimi-k2.7-code", + "date": "2026-06-20", + "value": "Listed on OpenRouter within days ($0.74/$3.50 via routed providers) with vLLM/SGLang self-hosting support; rides the mature K2-family ecosystem but is itself weeks old" + } + ], + "methodology": "Analysis of third-party hosting, integrations, and tooling; conservative given launch recency", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Modified MIT License", + "url": "https://huggingface.co/moonshotai/Kimi-K2.7-Code", + "date": "2026-06-12", + "value": "Same Modified MIT as K2.6: MIT with an attribution-UI requirement for deployments exceeding 100M MAU or $20M/month revenue" + } + ], + "methodology": "Review of licensing terms and restrictions; attribution clause is trust-relevant for large-scale commercial use", + "last_verified": "2026-07-09" + } + }, + "notes": "Inherits the K2 family's strong ecosystem (OpenRouter within days, Kimi Code CLI, vLLM/SGLang). Operational quirks to note: thinking mode cannot be disabled and sampling parameters are fixed server-side, which constrains tuning; the ~30% token-usage reduction partly offsets thinking-mode cost." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 94, + "notes": "Purpose-built for agentic coding on the strong K2.6 base with ~30% lower token usage; vendor claims large gains (62.0 Kimi Code Bench v2, 81.1 MCP Mark Verified vs Opus 4.8's 76.4) but all benchmarks are Moonshot-run — verify on your own workloads.", + "alternatives": ["kimi-k2-6", "claude-opus-4-8", "glm-5", "gpt-5-3-codex"] + }, + "customer-support": { + "overall": 72, + "notes": "Coding-specialized with mandatory thinking mode — poorly matched to simple support flows.", + "alternatives": ["command-a-plus", "minimax-m2"] + }, + "content-creation": { + "overall": 74, + "notes": "Not its purpose; the general-purpose K2.6 is the better Moonshot pick for prose.", + "alternatives": ["claude-opus-4-8", "kimi-k2-6"] + }, + "data-analysis": { + "overall": 85, + "notes": "Strong for code-heavy analysis pipelines (notebooks, ETL, tooling); K2.6 better for broad analytical reasoning.", + "alternatives": ["glm-5", "kimi-k2-6"] + }, + "research-assistant": { + "overall": 80, + "notes": "Capable tool-use and 256K context, but reasoning breadth trades toward code; use K2.6 for general research.", + "alternatives": ["kimi-k2-6", "glm-5"] + }, + "legal-compliance": { + "overall": 62, + "notes": "Wrong specialization, China-jurisdiction first-party API, and no Western certifications.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 60, + "notes": "Not recommended: coding-specialized and no compliant first-party path; health-software engineering should still self-host.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 76, + "notes": "Good for quant-developer workflows (code-first); data residency requires self-hosting for regulated firms.", + "alternatives": ["glm-5", "command-a-plus"] + }, + "education": { + "overall": 82, + "notes": "Strong coding tutor with visible thinking traces; less suited to general STEM tutoring than K2.6 or GLM-5.", + "alternatives": ["glm-5", "deepseek-v3-2"] + }, + "creative-writing": { + "overall": 70, + "notes": "Coding specialization comes at the cost of prose quality; pick a generalist.", + "alternatives": ["claude-opus-4-8", "minimax-m2"] + } + }, + + "strengths": [ + "Coding-specialized on the strong K2.6 base: vendor reports 62.0 Kimi Code Bench v2 (+21.8% over K2.6) and 81.1 MCP Mark Verified (vs Claude Opus 4.8's 76.4)", + "~30% lower reasoning-token usage than K2.6, cutting cost in agentic loops where thinking bills as output", + "Open weights under Modified MIT with full self-hosting (vLLM/SGLang) for keeping proprietary code in-house", + "Always-on thinking traces aid debugging and auditability of agent runs", + "256K context with text and image input (400M MoonViT encoder) for UI-aware coding", + "Broad availability within days: Kimi API, Kimi Code CLI, Hugging Face, OpenRouter ($0.74/$3.50 routed)" + ], + + "limitations": [ + "All published benchmarks are Moonshot-proprietary and vendor-run — no SWE-bench or independent public-suite results yet", + "Thinking mode cannot be disabled and sampling parameters are fixed server-side, limiting tuning", + "First-party Moonshot API processes data (including submitted source code) under Chinese jurisdiction with limited Western certifications", + "Modified MIT license imposes attribution-UI requirement above 100M MAU or $20M/month revenue", + "Coding specialization narrows general-purpose capability vs K2.6", + "Self-hosting a 1T-parameter MoE (~595 GB on disk) requires substantial GPU infrastructure", + "Weeks old: consistency, uptime, and security evidence still immature" + ], + + "best_for": [ + "Long-horizon agentic coding where reasoning-token spend dominates cost", + "Teams already on K2.6 or Kimi Code wanting a cheaper-per-task coding specialist", + "Self-hosted coding assistants that must keep proprietary source code in-house", + "Multi-turn repository engineering within a 256K context" + ], + + "not_recommended_for": [ + "Teams requiring independently verified benchmarks before adoption (vendor-only so far)", + "General-purpose or creative workloads — K2.6 or a generalist model fits better", + "Regulated Western workloads sending source code to the first-party API", + "Deployments needing custom sampling parameters or thinking-mode control", + "Hyperscale consumer products unwilling to satisfy the attribution-UI license clause" + ], + + "metadata": { + "pricing": { + "input": "$0.95 per 1M tokens ($0.19 cache hit)", + "output": "$4.00 per 1M tokens", + "notes": "First-party Moonshot API pricing at launch, same headline rates as K2.6; mandatory thinking mode bills as output, partly offset by ~30% lower token usage. OpenRouter routed pricing ~$0.74/$3.50.", + "last_verified": "2026-07-09" + }, + "context_window": 262144, + "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], + "modalities": ["text", "image (input)"], + "api_endpoint": "https://api.moonshot.ai/v1/chat/completions", + "open_source": true, + "license": "Modified MIT (attribution-UI requirement above 100M MAU or $20M/month revenue)", + "architecture": "Mixture-of-Experts built on Kimi K2.6: 1T total / 32B active parameters, 61 layers, 384 experts (8 selected + 1 shared), MLA attention, SwiGLU, 400M MoonViT vision encoder; coding-specialized post-training with mandatory thinking mode", + "parameters": "1T total / 32B active", + "release_date": "2026-06-12" + }, + + "related_entities": ["kimi-k2-6", "glm-5", "gpt-5-3-codex", "claude-opus-4-8", "minimax-m2"], + + "tags": [ + "coding", + "agentic", + "open-source", + "mixture-of-experts", + "long-context", + "token-efficient", + "chinese-provider", + "self-hostable" + ] +} diff --git a/data/models/minimax-m3.json b/data/models/minimax-m3.json new file mode 100644 index 0000000..b7aa66c --- /dev/null +++ b/data/models/minimax-m3.json @@ -0,0 +1,659 @@ +{ + "id": "minimax-m3", + "type": "model", + "name": "MiniMax-M3", + "provider": "MiniMax", + "version": "20260601", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "MiniMax's natively multimodal 428B MoE (23B active) with a 1M-token context via MiniMax Sparse Attention, launched June 2026 with weights on Hugging Face by 2026-06-07. Vendor reports 80.5% SWE-bench Verified; Artificial Analysis scores it 44 on Intelligence Index v4.1. Unlike MIT-licensed M2, M3 ships under a MiniMax Community License, and training code and some inference operators are withheld.", + "website": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + + "trust_vector": { + "performance_reliability": { + "overall_score": 88, + "criteria": { + "task_accuracy_code": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M3 Model Card (vendor-reported)", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "SWE-bench Verified 80.5%, SWE-bench Pro 59.0%, Terminal-Bench 2.1 66%, SWE-fficiency 34.8, KernelBench Hard 28.8" + }, + { + "source": "TechTimes launch analysis", + "url": "https://www.techtimes.com/articles/317532/20260601/minimax-m3-open-weight-coding-model-frontier-claims-unverified-benchmarks.htm", + "date": "2026-06-01", + "value": "Frontier claims flagged as vendor-run on MiniMax's own infrastructure with vendor-selected baselines" + } + ], + "methodology": "Vendor benchmarks with independent scrutiny noting all launch figures were vendor-run; partial independent corroboration via Artificial Analysis", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 85, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M3 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Reasoning modes (enabled/adaptive/disabled) via API; competitive reasoning for a 23B-active footprint" + } + ], + "methodology": "Vendor-reported reasoning benchmarks and early community evaluation", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis", + "url": "https://artificialanalysis.ai/models/minimax-m3", + "date": "2026-06-20", + "value": "Intelligence Index v4.1 score of 44 — well above the open-weight median (25), tied with DeepSeek V4 Pro, behind GLM-5.2 (51); roughly level with Claude Sonnet 4.6 on GDPval-AA agentic benchmark" + }, + { + "source": "MiniMax-M3 Model Card (multimodal)", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "MMMU Pro 78.1, Video-MME v2 85.4 — strong native multimodal understanding" + } + ], + "methodology": "Independent composite benchmarking (Artificial Analysis) plus vendor multimodal benchmarks", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 83, + "confidence": "low", + "evidence": [ + { + "source": "Community evaluation", + "url": "https://medium.com/@cognidownunder/i-evaluated-minimax-m3-for-agentic-workflows-the-results-are-complicated-518b60d5e6a9", + "date": "2026-06-25", + "value": "Early agentic evaluations report capable but mixed consistency across long workflows; limited observation window since June launch" + } + ], + "methodology": "Early community testing of repeated runs and agentic trajectories", + "last_verified": "2026-07-09" + }, + "latency_p50": { + "value": "1.9s", + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M3 Model Card (MSA)", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "MiniMax Sparse Attention delivers 9x prefill and 15x decode speedups vs M2 at 1M context, cutting per-token compute to 1/20; 23B active keeps generation fast" + } + ], + "methodology": "Median latency for API requests with standard prompt sizes; architecture-level efficiency verified in vendor technical report (arXiv:2606.13392)", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "4.5s", + "confidence": "low", + "evidence": [ + { + "source": "Community benchmarking", + "url": "https://artificialanalysis.ai/models/minimax-m3", + "date": "2026-06-20", + "value": "p95 ~4.5s; very long multimodal or 1M-context requests run longer" + } + ], + "methodology": "95th percentile response time from early third-party measurements", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "MiniMax-M3 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "1M-token context window enabled by MiniMax Sparse Attention" + } + ], + "methodology": "Official specification from model card", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 92, + "confidence": "low", + "evidence": [ + { + "source": "MiniMax Platform", + "url": "https://platform.minimax.io/", + "date": "2026-07-01", + "value": "First-party API stable in first weeks, with a paid Priority tier (1.5x) for admission and latency guarantees; no long-run availability record yet" + } + ], + "methodology": "Review of platform availability since launch; observation window under six weeks", + "last_verified": "2026-07-09" + } + }, + "notes": "Meaningful step over M2: 1M context, native multimodality, and independently confirmed intelligence gains (Artificial Analysis 44 vs open-weight median 25). Headline coding figures (80.5% SWE-bench Verified) are vendor-run on vendor infrastructure and drew explicit independent skepticism at launch; GLM-5.2 leads it on independent indices." + }, + + "security": { + "overall_score": 76, + "criteria": { + "prompt_injection_resistance": { + "score": 75, + "confidence": "low", + "evidence": [ + { + "source": "MiniMax-M3 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Safety tuning described; no published third-party prompt-injection audit; multimodal inputs (image/video) widen the injection surface" + } + ], + "methodology": "Review of vendor documentation against OWASP LLM01 patterns; model too new for mature red-team coverage", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "Community red-teaming", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-20", + "value": "Standard alignment tuning; open weights allow guardrail removal in derivatives; limited adversarial testing published since launch" + } + ], + "methodology": "Early testing against adversarial prompt datasets; deployer-dependent for self-hosted use", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Privacy Policy", + "url": "https://www.minimax.io/", + "date": "2026-06-01", + "value": "Standard data handling on first-party API; full control when self-hosted" + } + ], + "methodology": "Analysis of privacy policies and self-hosting data-control options", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 77, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M3 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Safety post-training applied; refusal behavior in line with peer open models per early reports" + } + ], + "methodology": "Safety testing across harmful content categories per vendor card and early community reports", + "last_verified": "2026-07-09" + }, + "api_security": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax API Documentation", + "url": "https://platform.minimax.io/docs", + "date": "2026-06-01", + "value": "API key authentication, HTTPS only, rate limiting; OpenAI- and Anthropic-compatible endpoints carried over from M2" + } + ], + "methodology": "Review of API security features and best practices", + "last_verified": "2026-07-09" + } + }, + "notes": "Standard open-model posture without third-party audits, scored slightly below M2 for launch recency and the wider multimodal attack surface. Self-hosting shifts security responsibility to the deployer." + }, + + "privacy_compliance": { + "overall_score": 73, + "criteria": { + "data_residency": { + "value": "China (first-party MiniMax API); any jurisdiction when self-hosted or via Western hosts", + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Platform Documentation", + "url": "https://platform.minimax.io/docs", + "date": "2026-06-01", + "value": "MiniMax is a China-based provider; first-party API data processed under Chinese jurisdiction" + }, + { + "source": "OpenRouter availability", + "url": "https://openrouter.ai/minimax/minimax-m3", + "date": "2026-06-20", + "value": "Open weights served by Western inference providers, enabling non-China residency (subject to Community License terms)" + } + ], + "methodology": "Review of provider jurisdiction and third-party hosting options", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Privacy Policy", + "url": "https://www.minimax.io/", + "date": "2026-06-01", + "value": "Standard API data terms; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of privacy policy and data usage terms", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "Per MiniMax policy on first-party API (China jurisdiction); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Terms of Service", + "url": "https://www.minimax.io/", + "date": "2026-06-01", + "value": "First-party retention governed by Chinese data regulations; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of terms of service and deployment-dependent retention", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 70, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Documentation", + "url": "https://platform.minimax.io/docs", + "date": "2026-06-01", + "value": "Customer responsible for PII redaction; no managed PII tooling; multimodal inputs (images/video) add PII-handling complexity" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 66, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax public materials", + "url": "https://www.minimax.io/", + "date": "2026-06-01", + "value": "No published SOC 2 / HIPAA / GDPR attestations for the first-party API" + } + ], + "methodology": "Verification of compliance certifications and audit reports", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Open weights on Hugging Face", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Self-hosting (vLLM/SGLang/Transformers/KTransformers) gives complete data control and zero external retention; Community License conditions apply to commercial deployments" + } + ], + "methodology": "Review of self-hosting deployment options enabling zero retention", + "last_verified": "2026-07-09" + } + }, + "notes": "First-party MiniMax API operates under Chinese jurisdiction — a material caveat for Western regulated industries. Self-hosting mitigates residency, but unlike MIT-licensed M2, M3's Community License attaches attribution and revenue-threshold conditions to commercial self-hosted use." + }, + + "trust_transparency": { + "overall_score": 75, + "criteria": { + "explainability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M3 documentation", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Reasoning modes (enabled/adaptive/disabled) expose or suppress thinking traces per request, aiding agent-loop auditability" + } + ], + "methodology": "Evaluation of reasoning transparency and trajectory inspectability", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 76, + "confidence": "low", + "evidence": [ + { + "source": "Community testing", + "url": "https://medium.com/@cognidownunder/i-evaluated-minimax-m3-for-agentic-workflows-the-results-are-complicated-518b60d5e6a9", + "date": "2026-06-25", + "value": "Early agentic evaluations report moderate hallucination; tool-grounded workflows perform better than closed-book QA" + } + ], + "methodology": "Early testing on factual QA and tool-augmented workflows", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 72, + "confidence": "low", + "evidence": [ + { + "source": "MiniMax-M3 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Limited published bias evaluation, including for multimodal inputs" + } + ], + "methodology": "Review of published bias benchmarks and community evaluations", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 74, + "confidence": "low", + "evidence": [ + { + "source": "Model behavior testing", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-20", + "value": "Basic uncertainty expression; no calibrated confidence outputs; limited testing since launch" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 84, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face model card and MSA technical report", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Clear documentation of 428B/23B MoE, MSA architecture (arXiv:2606.13392), benchmark tables, recommended inference parameters, and deployment guidance" + } + ], + "methodology": "Review of documentation completeness and clarity", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 64, + "confidence": "medium", + "evidence": [ + { + "source": "Independent launch analysis", + "url": "https://codersera.com/blog/minimax-m3-release-date-whats-new-2026/", + "date": "2026-06-10", + "value": "Training code, data pipelines, and specific inference operators withheld; data mixture not published — a step back from full open-source transparency" + }, + { + "source": "Kili Technology data story", + "url": "https://kili-technology.com/blog/data-story-minimax-m3", + "date": "2026-06-15", + "value": "Analysis of what MiniMax disclosed vs withheld about M3's training data and pipeline" + } + ], + "methodology": "Review of public disclosures about training data and released artifacts", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax-M3 Model Card", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3", + "date": "2026-06-07", + "value": "Built-in safety tuning; deployers of open weights must layer their own guardrails, including for image/video inputs" + } + ], + "methodology": "Analysis of built-in safety mechanisms", + "last_verified": "2026-07-09" + } + }, + "notes": "Good architectural documentation (MSA technical report) but reduced openness vs M2: training code, data pipelines, and some inference operators are withheld, and all launch benchmarks were vendor-run — independent reviewers explicitly flagged the verification gap." + }, + + "operational_excellence": { + "overall_score": 77, + "criteria": { + "api_design_quality": { + "score": 85, + "confidence": "high", + "evidence": [ + { + "source": "MiniMax API Documentation", + "url": "https://platform.minimax.io/docs/guides/pricing-paygo", + "date": "2026-06-01", + "value": "OpenAI- and Anthropic-compatible endpoints with streaming, tool calling, reasoning modes, prompt caching, and a Priority service tier" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax GitHub", + "url": "https://github.com/MiniMax-AI", + "date": "2026-06-07", + "value": "Compatibility with mainstream OpenAI/Anthropic SDKs; first-party tooling adequate but M3-specific tooling still young" + } + ], + "methodology": "Review of SDK quality, documentation, and maintenance", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax release history", + "url": "https://huggingface.co/MiniMaxAI", + "date": "2026-06-07", + "value": "M3 announced 2026-06-01 with weights on Hugging Face by 2026-06-07; M2 weights remain available; license changed from MIT (M2) to MiniMax Community License (M3) between generations" + } + ], + "methodology": "Review of versioning practices, weight availability, and license continuity across releases", + "last_verified": "2026-07-09" + }, + "monitoring_observability": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax Platform", + "url": "https://platform.minimax.io/", + "date": "2026-06-01", + "value": "Basic usage dashboard; self-hosted observability is deployer-built" + } + ], + "methodology": "Review of available monitoring tools and metrics", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 76, + "confidence": "medium", + "evidence": [ + { + "source": "MiniMax community channels", + "url": "https://github.com/MiniMax-AI", + "date": "2026-06-01", + "value": "GitHub and community support; limited English-language enterprise support" + } + ], + "methodology": "Assessment of documentation, community, and support responsiveness", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 79, + "confidence": "medium", + "evidence": [ + { + "source": "Inference ecosystem", + "url": "https://openrouter.ai/minimax/minimax-m3", + "date": "2026-06-20", + "value": "Day-one vLLM/SGLang/Transformers/KTransformers/unsloth support with community quantizations and OpenRouter availability; ~259K Hugging Face downloads in the first month, though the M3-specific ecosystem is only weeks old" + } + ], + "methodology": "Analysis of third-party hosting, integrations, and tooling; conservative given launch recency", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 70, + "confidence": "high", + "evidence": [ + { + "source": "MiniMax Community License", + "url": "https://huggingface.co/MiniMaxAI/MiniMax-M3/blob/main/LICENSE", + "date": "2026-06-07", + "value": "Not MIT/Apache: commercial use requires a visible 'Built with MiniMax M3' notice, and products exceeding the license's yearly revenue threshold need separate authorization" + }, + { + "source": "Independent license analysis", + "url": "https://www.verdent.ai/guides/how-to-use-minimax-m3", + "date": "2026-06-15", + "value": "'Available to download' does not mean 'free for any commercial use' — reviewers advise reading the license before building commercial products on M3" + } + ], + "methodology": "Review of licensing terms and restrictions; regression from M2's clean MIT license is trust-relevant", + "last_verified": "2026-07-09" + } + }, + "notes": "Dual OpenAI/Anthropic compatibility and broad day-one inference support carry over from M2, but the license regression (MIT to MiniMax Community License with attribution and revenue-threshold clauses) and withheld training/inference code lower the operational score. Commercial deployers need legal review." + } + }, + + "use_case_ratings": { + "code-generation": { + "overall": 88, + "notes": "Vendor-reported 80.5% SWE-bench Verified and 59.0% SWE-bench Pro at very low cost, but figures are vendor-run and GLM-5.2 leads on independent indices.", + "alternatives": ["glm-5", "kimi-k2-6"] + }, + "customer-support": { + "overall": 85, + "notes": "Fast (23B active), very cheap, and multimodal — can handle screenshot- and image-based support tickets natively.", + "alternatives": ["command-a-plus", "minimax-m2"] + }, + "content-creation": { + "overall": 82, + "notes": "Good generation quality at minimal cost; image/video understanding aids content workflows.", + "alternatives": ["glm-5", "claude-opus-4-8"] + }, + "data-analysis": { + "overall": 85, + "notes": "Good tool calling plus native chart/image understanding; weaker raw reasoning than GLM-5.2 or Kimi K2.6.", + "alternatives": ["glm-5", "kimi-k2-6"] + }, + "research-assistant": { + "overall": 87, + "notes": "1M context and native multimodality (text/image/video) suit long-document and mixed-media research at low cost.", + "alternatives": ["glm-5", "kimi-k2-6"] + }, + "legal-compliance": { + "overall": 66, + "notes": "China-jurisdiction first-party API, no Western certifications, and Community License conditions make this a poor fit without careful mitigation.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "healthcare": { + "overall": 65, + "notes": "Not recommended via first-party API; self-hosted deployment in a compliant environment is the only viable path, with license review.", + "alternatives": ["command-a-plus", "claude-opus-4-8"] + }, + "financial-analysis": { + "overall": 80, + "notes": "Capable and cheap with chart/document understanding; data residency requires self-hosting for regulated firms.", + "alternatives": ["command-a-plus", "glm-5"] + }, + "education": { + "overall": 83, + "notes": "Multimodal tutoring (diagrams, video) at very low cost suits high-volume educational platforms.", + "alternatives": ["glm-5", "deepseek-v3-2"] + }, + "creative-writing": { + "overall": 78, + "notes": "Serviceable creative output; not its design focus.", + "alternatives": ["claude-opus-4-8", "glm-5"] + } + }, + + "strengths": [ + "Native multimodality: mixed-modality training from step one across text, image, and video (78.1 MMMU Pro, 85.4 Video-MME v2)", + "1M-token context with MiniMax Sparse Attention: 9x prefill / 15x decode speedups vs M2 at 1M context", + "Efficient 23B-active footprint keeps inference fast and self-hosting relatively affordable", + "Aggressive pricing: $0.30/$1.20 per 1M tokens (promotional 50% off list) with $0.06 cache reads", + "Independently confirmed intelligence gain: Artificial Analysis 44, well above the open-weight median", + "Day-one vLLM/SGLang/Transformers/KTransformers/unsloth support with community quantizations" + ], + + "limitations": [ + "License regression from M2: MiniMax Community License requires 'Built with MiniMax M3' attribution and separate authorization above a revenue threshold", + "Training code, data pipelines, and some inference operators withheld despite the 'open-weight' label", + "All launch benchmarks vendor-run on vendor infrastructure; independent reviewers flagged the verification gap", + "First-party API processes data under Chinese jurisdiction with no published Western compliance certifications", + "Behind GLM-5.2 on independent intelligence and agentic indices", + "Long-context pricing doubles above 512K input tokens", + "Launched June 2026 — consistency, uptime, and security evidence still immature" + ], + + "best_for": [ + "Multimodal agentic workloads mixing text, screenshots, documents, and video at low cost", + "1M-context document and codebase analysis on a budget", + "High-volume workloads where the 23B-active footprint's speed and price dominate", + "Teams already on M2 wanting multimodality and longer context with API compatibility" + ], + + "not_recommended_for": [ + "Regulated Western workloads (healthcare, legal, finance) on the first-party API", + "Commercial products unwilling to carry 'Built with MiniMax M3' attribution or negotiate above the revenue threshold", + "Teams requiring vendor-independent benchmark verification before adoption", + "Workloads needing the strongest open-weight coding performance (GLM-5.2 leads independently)" + ], + + "metadata": { + "pricing": { + "input": "$0.30 per 1M tokens (≤512K input; $0.06 cache hit)", + "output": "$1.20 per 1M tokens", + "notes": "Promotional 'permanent 50% off' from list $0.60/$2.40. Rates double above 512K input tokens; Priority tier is 1.5x for admission/latency guarantees. Third-party host pricing varies.", + "last_verified": "2026-07-09" + }, + "context_window": 1000000, + "languages": ["English", "Chinese", "Japanese", "Korean", "Spanish", "French", "German"], + "modalities": ["text", "image (input)", "video (input)"], + "api_endpoint": "https://api.minimax.io/v1/chat/completions", + "open_source": true, + "license": "MiniMax Community License (attribution notice required for commercial use; separate authorization above yearly revenue threshold)", + "architecture": "Mixture-of-Experts: ~428B total / ~23B active parameters, MiniMax Sparse Attention (MSA), native mixed-modality training (arXiv:2606.13392)", + "parameters": "428B total / 23B active", + "release_date": "2026-06-01" + }, + + "related_entities": ["minimax-m2", "glm-5", "kimi-k2-6", "deepseek-v4", "qwen3-5"], + + "tags": [ + "agentic", + "multimodal", + "tool-calling", + "open-weight", + "mixture-of-experts", + "long-context", + "cost-effective", + "fast-inference", + "chinese-provider", + "self-hostable" + ] +} diff --git a/data/models/nemotron-3-ultra.json b/data/models/nemotron-3-ultra.json new file mode 100644 index 0000000..0a71b5b --- /dev/null +++ b/data/models/nemotron-3-ultra.json @@ -0,0 +1,684 @@ +{ + "id": "nemotron-3-ultra", + "type": "model", + "name": "Nemotron 3 Ultra", + "provider": "NVIDIA", + "version": "20260604", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "NVIDIA's open frontier reasoning model, released 2026-06-04 to complete the Nemotron 3 rollout (Nano Dec 2025, Super Mar 2026): a 550B total / 55B active LatentMoE hybrid Mamba-Transformer under OpenMDW-1.1 with open weights, training data, and recipes. 1M-token context, 71.9% SWE-bench Verified (vendor), Artificial Analysis Index 48 — the top-scoring US open-weight model — with ~140 tok/s decode and the best non-hallucination score in its comparison set (78.7 AA-Omniscience).", + "website": "https://research.nvidia.com/labs/nemotron/Nemotron-3-Ultra/", + "trust_vector": { + "performance_reliability": { + "overall_score": 91, + "criteria": { + "task_accuracy_code": { + "score": 94, + "confidence": "medium", + "evidence": [ + { + "source": "MarkTechPost - Nemotron 3 Ultra release coverage", + "url": "https://www.marktechpost.com/2026/06/04/nvidia-ai-releases-nemotron-3-ultra-an-open-550b-mixture-of-experts-hybrid-mamba-transformer-for-long-running-agents/", + "date": "2026-06-04", + "value": "SWE-bench Verified 71.9, IOI 2025 570.0 (vendor-reported)" + }, + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "SWE-bench Verified 65.0-70.4 across five agent harnesses in independent runs, slightly below the 71.9 vendor peak; Terminal-Bench 2.1 56.4 (vendor-stated)" + } + ], + "methodology": "Vendor coding benchmarks cross-checked against independent multi-harness SWE-bench reproductions", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "Artificial Analysis Intelligence Index (via Digital Applied)", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "Intelligence Index 48 (#9 of 89): highest-scoring US open-weight model, 6 points behind open leader Kimi K2.6 (54)" + }, + { + "source": "MarkTechPost - Nemotron 3 Ultra release coverage", + "url": "https://www.marktechpost.com/2026/06/04/nvidia-ai-releases-nemotron-3-ultra-an-open-550b-mixture-of-experts-hybrid-mamba-transformer-for-long-running-agents/", + "date": "2026-06-04", + "value": "PinchBench 90.0 held-out; post-trained with SFT, RLVR, and Multi-teacher On-Policy Distillation for long-running agent reasoning" + } + ], + "methodology": "Independent aggregate intelligence index plus vendor reasoning benchmarks; some vendor numbers await third-party replication", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 90, + "confidence": "medium", + "evidence": [ + { + "source": "MarkTechPost - Nemotron 3 Ultra release coverage", + "url": "https://www.marktechpost.com/2026/06/04/nvidia-ai-releases-nemotron-3-ultra-an-open-550b-mixture-of-experts-hybrid-mamba-transformer-for-long-running-agents/", + "date": "2026-06-04", + "value": "AA-Omniscience 78.7, the highest non-hallucination score in its comparison set; RULER 94.7 at 1M tokens" + } + ], + "methodology": "Knowledge and long-context retrieval benchmarks from launch materials, corroborated by independent press analysis", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 86, + "confidence": "low", + "evidence": [ + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "Stable results across agent harnesses but notably verbose: ~2.3x more output tokens than peer models" + } + ], + "methodology": "Cross-harness benchmark variance and community reports; only one month of production usage available", + "last_verified": "2026-07-09", + "notes": "Launch-recency: limited repeated-run data from the community so far" + }, + "latency_p50": { + "value": "~1.3s time-to-first-token; ~140 tokens/s decode (hosted)", + "confidence": "medium", + "evidence": [ + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "140.3 tokens/s output (#7 of 89 models) with 1.33s time-to-first-token; several times faster than Kimi K2.6 (50-100 tok/s)" + }, + { + "source": "MarkTechPost - Nemotron 3 Ultra release coverage", + "url": "https://www.marktechpost.com/2026/06/04/nvidia-ai-releases-nemotron-3-ultra-an-open-550b-mixture-of-experts-hybrid-mamba-transformer-for-long-running-agents/", + "date": "2026-06-04", + "value": "5.9x throughput vs GLM-5.1 and 4.8x vs Kimi K2.6 at 8K input / 64K output (vendor)" + } + ], + "methodology": "Independent hosted-endpoint speed measurements plus vendor throughput comparisons", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "Provider-dependent; fast decode limits tail latency, but reasoning verbosity (~2.3x peer token counts) lengthens end-to-end task time", + "confidence": "low", + "evidence": [ + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "High decode speed offset by 2.3x output-token verbosity on reasoning tasks; per-provider tail latencies not yet broadly characterized" + } + ], + "methodology": "Qualitative assessment from early independent benchmarking; p95 distributions not yet published one month post-launch", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "1,000,000 tokens", + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "1M-token context window; RULER 94.7 at 1M tokens (some hosts serve reduced limits)" + } + ], + "methodology": "Official specification from model card; note that individual API providers may cap served context", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "OpenRouter model listing", + "url": "https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b", + "date": "2026-07-09", + "value": "Day-zero availability across 25+ platforms (NVIDIA NIM, OpenRouter, Together AI, Fireworks, Perplexity, SageMaker JumpStart) plus self-hosting redundancy" + } + ], + "methodology": "Multi-provider availability assessment; only one month of hosted operating history", + "last_verified": "2026-07-09" + } + }, + "notes": "Strongest US open-weight model on the Artificial Analysis Intelligence Index (48, #9 overall) with class-leading decode speed (~140 tok/s) and a 1M-token context. Vendor benchmark peaks (SWE-bench 71.9) run a few points above independent multi-harness reproductions (65.0-70.4). Verbosity (~2.3x peer output tokens) is the main efficiency caveat." + }, + "security": { + "overall_score": 82, + "criteria": { + "prompt_injection_resistance": { + "score": 80, + "confidence": "low", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "Safety post-training documented; no published third-party injection red-team results one month post-launch" + } + ], + "methodology": "Review of vendor safety documentation; independent OWASP LLM01 testing not yet available for this release", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 80, + "confidence": "low", + "evidence": [ + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "NVIDIA ships a companion Nemotron 3.5 Content Safety model (4B, 23 safety categories, 12 languages, released 2026-06-02); base alignment is removable downstream given open weights" + } + ], + "methodology": "Assessment of vendor guardrail stack; accounts for open-weight modifiability and absence of independent adversarial evaluations", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "Open weights allow fully self-hosted deployment with complete data control; hosted access follows each provider's data handling terms" + } + ], + "methodology": "Analysis of deployment options: self-hosting gives full data isolation, hosted routes depend on provider policies", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "NVIDIA Nemotron research page", + "url": "https://research.nvidia.com/labs/nemotron/Nemotron-3-Ultra/", + "date": "2026-06-04", + "value": "Safety-aligned release with documented post-training; pairs with the Nemotron 3.5 Content Safety classifier covering 23 categories across 12 languages" + } + ], + "methodology": "Review of default-weight safety behavior and the vendor's companion moderation stack", + "last_verified": "2026-07-09" + }, + "api_security": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "OpenRouter model listing", + "url": "https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b", + "date": "2026-07-09", + "value": "NVIDIA NIM and major hosts provide API-key authentication, HTTPS, and rate limiting via OpenAI-compatible endpoints" + } + ], + "methodology": "Review of API security features on NVIDIA NIM and principal third-party hosts", + "last_verified": "2026-07-09" + } + }, + "notes": "Reasonable default safety stack including a dedicated companion content-safety model, but almost no independent red-team literature exists yet given the June 2026 launch. Open weights shift guardrail responsibility to deployers who fine-tune." + }, + "privacy_compliance": { + "overall_score": 83, + "criteria": { + "data_residency": { + "value": "Deployment-dependent: US-jurisdiction NVIDIA NIM, 25+ third-party hosts, or self-hosted in any region", + "confidence": "high", + "evidence": [ + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "Day-zero availability on OpenRouter, NVIDIA NIM, Together AI, Fireworks, Perplexity, and Amazon SageMaker JumpStart; OpenMDW weights allow deployment anywhere" + } + ], + "methodology": "Review of hosting options and open-weight licensing; residency is fully controllable via self-hosting", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "NVIDIA Privacy Policy", + "url": "https://www.nvidia.com/en-us/privacy/", + "date": "2026-06-04", + "value": "NVIDIA hosted services do not train on API data by default; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of NVIDIA hosted-service terms plus the self-hosting option", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "Per host policy on managed endpoints; zero external retention when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "NVIDIA Terms of Service", + "url": "https://www.nvidia.com/en-us/terms/", + "date": "2026-06-04", + "value": "Hosted retention follows NVIDIA or third-party host policies; open weights make retention fully deployment-dependent" + } + ], + "methodology": "Review of hosted-platform retention policies across the deployment spectrum", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "Customer responsible for PII redaction; NVIDIA documents data provenance and provides NeMo Guardrails tooling in the surrounding stack" + } + ], + "methodology": "Review of data protection capabilities and deployer responsibilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "NVIDIA Trust Center", + "url": "https://www.nvidia.com/en-us/trust/", + "date": "2026-06-04", + "value": "NVIDIA infrastructure holds SOC 2 Type II and GDPR programs; no model-service HIPAA/FedRAMP path — regulated deployments inherit certifications from the chosen host (e.g., SageMaker) or self-hosted environment" + } + ], + "methodology": "Verification of provider infrastructure certifications versus model-service-level compliance", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "No formal zero-retention guarantee on hosted endpoints; self-hosting provides true zero external retention" + } + ], + "methodology": "Review of data handling across NVIDIA NIM, third-party hosts, and self-hosting", + "last_verified": "2026-07-09" + } + }, + "notes": "US-jurisdiction provider with unusually flexible residency thanks to open weights and 25+ hosts, including hyperscaler routes (SageMaker JumpStart) that carry their own certifications. No model-level HIPAA/FedRAMP; regulated buyers should deploy via certified infrastructure." + }, + "trust_transparency": { + "overall_score": 85, + "criteria": { + "explainability": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "MarkTechPost - Nemotron 3 Ultra release coverage", + "url": "https://www.marktechpost.com/2026/06/04/nvidia-ai-releases-nemotron-3-ultra-an-open-550b-mixture-of-experts-hybrid-mamba-transformer-for-long-running-agents/", + "date": "2026-06-04", + "value": "Reasoning model exposing chain-of-thought traces; fully inspectable weights and recipes when self-hosted" + } + ], + "methodology": "Evaluation of reasoning-trace accessibility and weight/recipe inspectability", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 88, + "confidence": "medium", + "evidence": [ + { + "source": "MarkTechPost - Nemotron 3 Ultra release coverage", + "url": "https://www.marktechpost.com/2026/06/04/nvidia-ai-releases-nemotron-3-ultra-an-open-550b-mixture-of-experts-hybrid-mamba-transformer-for-long-running-agents/", + "date": "2026-06-04", + "value": "78.7 on AA-Omniscience — the highest non-hallucination score in its comparison set" + } + ], + "methodology": "Non-hallucination benchmark results from launch materials, pending broader independent replication", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 78, + "confidence": "low", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "NVIDIA Responsible AI documentation and data provenance disclosed; independent bias audits not yet published for this release" + } + ], + "methodology": "Review of vendor responsible-AI disclosures; third-party fairness evaluations pending given launch recency", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "low", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "Expresses uncertainty within reasoning traces; calibration data not yet independently characterized" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 92, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Nemotron v3 collection", + "url": "https://huggingface.co/collections/nvidia/nvidia-nemotron-v3", + "date": "2026-06-04", + "value": "Detailed cards for BF16, Base-BF16, NVFP4, and GenRM variants covering architecture (LatentMoE, MTP), training phases, data cutoffs (pre-train Sep 2025, post-train May 2026), and deployment recipes" + } + ], + "methodology": "Review of model card and technical documentation completeness", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 94, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Base model card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-Base-BF16", + "date": "2026-06-04", + "value": "Openly released training corpora: 20T-token pre-training mix documented (Nemotron-CC, Common Crawl snapshots CC-MAIN-2013-20 through 2025-13), plus 10M SFT samples and 1M RL tasks published under OpenMDW-1.1" + } + ], + "methodology": "Review of published training datasets and recipes; best-in-class disclosure among frontier-scale models", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "Safety alignment in released weights plus the companion Nemotron 3.5 Content Safety 4B classifier; removable by downstream fine-tuning" + } + ], + "methodology": "Analysis of built-in safety mechanisms in default weights and companion tooling", + "last_verified": "2026-07-09" + } + }, + "notes": "Transparency is this release's standout: weights, 20T-token training data, and recipes are all published under OpenMDW-1.1 — materially beyond typical open-weight disclosure. Independent bias and calibration audits are still pending one month post-launch." + }, + "operational_excellence": { + "overall_score": 85, + "criteria": { + "api_design_quality": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "OpenRouter model listing", + "url": "https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b", + "date": "2026-07-09", + "value": "OpenAI-compatible endpoints via NVIDIA NIM and 25+ hosts with function calling and reasoning controls" + } + ], + "methodology": "Review of API design and feature completeness across NIM and principal hosts", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 87, + "confidence": "high", + "evidence": [ + { + "source": "Digital Applied independent analysis", + "url": "https://www.digitalapplied.com/blog/nvidia-nemotron-3-ultra-550b-open-reasoning-model-2026", + "date": "2026-06-05", + "value": "Day-zero support in vLLM, SGLang, and TensorRT-LLM; Hugging Face Transformers integration; single NVFP4 checkpoint spans Ampere through Blackwell" + } + ], + "methodology": "Review of inference-framework support and deployment tooling at launch", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "NVIDIA Nemotron 3 family announcement", + "url": "https://nvidianews.nvidia.com/news/nvidia-debuts-nemotron-3-family-of-open-models", + "date": "2025-12-15", + "value": "Predictable staged rollout: Nemotron 3 Nano (Dec 2025), Super 120B-A12B (2026-03-11), Ultra 550B-A55B (2026-06-04)" + }, + { + "source": "NVIDIA Nemotron research page", + "url": "https://research.nvidia.com/labs/nemotron/Nemotron-3-Ultra/", + "date": "2026-07-09", + "value": "Open weights remain permanently downloadable, softening deprecation risk; note NVIDIA retired the prior Llama-3.1-based Nemotron line within about a year" + } + ], + "methodology": "Review of release cadence, family roadmap execution, and weight-availability guarantees", + "last_verified": "2026-07-09" + }, + "monitoring_observability": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "OpenRouter model listing", + "url": "https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b", + "date": "2026-07-09", + "value": "Usage dashboards on NIM and third-party hosts; full observability when self-hosting on vLLM/SGLang/TRT-LLM stacks" + } + ], + "methodology": "Review of monitoring tools across deployment options", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "NVIDIA AI Enterprise", + "url": "https://www.nvidia.com/en-us/ai/enterprise/", + "date": "2026-06-04", + "value": "Enterprise support with SLAs available via NVIDIA AI Enterprise; strong developer documentation and community channels" + } + ], + "methodology": "Assessment of enterprise support tiers, documentation, and community responsiveness", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 80, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Nemotron v3 collection", + "url": "https://huggingface.co/collections/nvidia/nvidia-nemotron-v3", + "date": "2026-07-09", + "value": "Complete family ladder (Nano 30B-A3B, Super 120B-A12B, Ultra 550B-A55B) with early community quantizations (e.g., Unsloth GGUF); derivative ecosystem still nascent one month post-launch" + } + ], + "methodology": "Analysis of derivative models, third-party hosting breadth, and tooling integrations; scored conservatively given launch recency", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card", + "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", + "date": "2026-06-04", + "value": "OpenMDW License 1.1 (Linux Foundation): permissive, covers weights, data, and recipes, with patent-termination clause; less legally familiar to enterprises than Apache 2.0" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-07-09" + } + }, + "notes": "Strong launch operations: day-zero support across 25+ hosts and all major inference frameworks, plus a single NVFP4 checkpoint spanning Ampere to Blackwell. Ecosystem and monitoring scores held conservative pending more production history; OpenMDW-1.1 is permissive but newer than Apache 2.0." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 91, + "notes": "71.9% SWE-bench Verified (vendor; 65.0-70.4 independent) with fast decode suits agentic coding; verbosity raises per-task output cost.", + "alternatives": [ + "kimi-k2-6", + "gpt-5-3-codex" + ] + }, + "customer-support": { + "overall": 81, + "notes": "Capable but text-only, verbose, and primarily English-optimized; oversized for most support tiers.", + "alternatives": [ + "claude-haiku-4-5", + "gemini-3-5-flash" + ] + }, + "content-creation": { + "overall": 82, + "notes": "Solid structured writing; reasoning verbosity requires prompt discipline for concise content.", + "alternatives": [ + "claude-opus-4-6", + "gpt-5-5" + ] + }, + "data-analysis": { + "overall": 89, + "notes": "1M-token context with RULER 94.7 handles very large datasets and logs; text-only, so charts/images need preprocessing.", + "alternatives": [ + "gemini-3-pro", + "claude-opus-4-6" + ] + }, + "research-assistant": { + "overall": 90, + "notes": "1M context plus the best non-hallucination score in its set (78.7 AA-Omniscience) suits long-document research and long-running agents.", + "alternatives": [ + "gemini-3-pro", + "claude-opus-4-6" + ] + }, + "legal-compliance": { + "overall": 79, + "notes": "Compliance is deployment-dependent: viable via certified hosts (e.g., SageMaker) or self-hosting; no model-service certifications.", + "alternatives": [ + "claude-opus-4-6" + ] + }, + "healthcare": { + "overall": 74, + "notes": "No HIPAA path on NVIDIA-hosted endpoints; deploy in compliant infrastructure via open weights.", + "alternatives": [ + "claude-opus-4-6" + ] + }, + "financial-analysis": { + "overall": 86, + "notes": "Strong reasoning and long-context document processing; independent calibration data still limited.", + "alternatives": [ + "gpt-5-5", + "claude-opus-4-6" + ] + }, + "education": { + "overall": 83, + "notes": "Clear reasoning traces aid tutoring, but verbosity and English-first coverage limit fit versus multilingual alternatives.", + "alternatives": [ + "qwen3-5", + "gemini-3-flash" + ] + }, + "creative-writing": { + "overall": 78, + "notes": "Reasoning-optimized rather than prose-optimized; competent but not distinctive creative output.", + "alternatives": [ + "claude-opus-4-6", + "gpt-5-5" + ] + } + }, + "strengths": [ + "Top-scoring US open-weight model: Artificial Analysis Intelligence Index 48 (#9 of 89)", + "Radical transparency: weights, 20T-token training data, and recipes all released under OpenMDW-1.1", + "1M-token context with RULER 94.7 at full length", + "Fast for its class: ~140 tok/s decode, 1.33s TTFT; vendor-reported 4.8-5.9x throughput vs open peers", + "Best non-hallucination score in its comparison set (78.7 AA-Omniscience)", + "Single NVFP4 checkpoint runs across Ampere, Hopper, and Blackwell; day-zero vLLM/SGLang/TRT-LLM support", + "Completes a predictable family ladder (Nano 30B-A3B, Super 120B-A12B, Ultra 550B-A55B)" + ], + "limitations": [ + "Very verbose: ~2.3x more output tokens than peers, inflating cost and end-to-end latency on reasoning tasks", + "Trails the open-weight leader Kimi K2.6 by 6 Intelligence Index points (48 vs 54)", + "Vendor benchmark peaks exceed independent reproductions (SWE-bench 71.9 vendor vs 65.0-70.4 across harnesses)", + "Text-only: no image, audio, or video input", + "Launched 2026-06-04: minimal independent red-teaming, bias audits, or production track record", + "550B total parameters require serious multi-GPU infrastructure to self-host", + "OpenMDW-1.1 license is permissive but less familiar to enterprise legal teams than Apache 2.0" + ], + "best_for": [ + "Long-running autonomous agents needing 1M-token context and low hallucination", + "Organizations requiring full-stack auditability (open weights, data, and recipes)", + "Agentic coding on US-jurisdiction open infrastructure", + "NVIDIA-based self-hosting from Ampere to Blackwell via one NVFP4 checkpoint", + "Long-document analysis and retrieval at extreme context lengths" + ], + "not_recommended_for": [ + "Multimodal applications (image/audio/video input)", + "Cost-sensitive workloads where output-token verbosity dominates spend", + "Regulated deployments via NVIDIA-hosted endpoints alone (no HIPAA/FedRAMP model-service path)", + "Teams without GPU infrastructure that also need on-prem deployment", + "High-assurance uses requiring an established independent security track record" + ], + "metadata": { + "pricing": { + "input": "Free weights (OpenMDW-1.1); hosted ~$0.50 per 1M tokens (OpenRouter reference)", + "output": "Hosted ~$2.20 per 1M tokens (OpenRouter reference); free-tier variant available", + "notes": "Self-hosting is infrastructure-cost-only. Budget for ~2.3x peer output-token verbosity on reasoning tasks, which erodes headline per-token savings. Rates vary by host.", + "last_verified": "2026-07-09" + }, + "context_window": 1000000, + "languages": [ + "English", + "German", + "Spanish", + "French", + "Italian", + "Japanese" + ], + "modalities": [ + "text" + ], + "api_endpoint": "https://integrate.api.nvidia.com/v1", + "open_source": true, + "architecture": "LatentMoE hybrid Mamba-Transformer: interleaved Mamba-2, MoE (512 experts, top-22 active), and select attention layers across 108 layers, with Multi-Token Prediction and NVFP4 4-bit pretraining", + "parameters": "550B total / 55B active" + }, + "related_entities": [ + "nemotron-ultra-253b", + "kimi-k2-6", + "glm-5", + "qwen3-5", + "deepseek-v4" + ], + "tags": [ + "open-source", + "mixture-of-experts", + "hybrid-mamba-transformer", + "long-context", + "reasoning", + "agentic", + "open-training-data", + "nvidia-ecosystem", + "flagship" + ] +} diff --git a/data/models/qwen3-6.json b/data/models/qwen3-6.json new file mode 100644 index 0000000..8dfc887 --- /dev/null +++ b/data/models/qwen3-6.json @@ -0,0 +1,652 @@ +{ + "id": "qwen3-6", + "type": "model", + "name": "Qwen3.6", + "provider": "Alibaba", + "version": "20260416", + "last_evaluated": "2026-07-09", + "evaluated_by": "TrustVector Team", + "description": "Alibaba's Apache-2.0 open-weight Qwen3.6 family (Apr 2026): Qwen3.6-35B-A3B MoE (35B total / 3B active, 2026-04-16) and Qwen3.6-27B dense (2026-04-22). Hybrid Gated DeltaNet + Gated Attention with thinking mode and Thinking Preservation; 262K native context (~1M via YaRN); text, image, and video input across 201 languages. The 27B beats the 397B-A17B Qwen3.5 flagship on agentic coding (77.2% SWE-bench Verified). The newer Qwen3.7-Max/Plus frontier remains API-only.", + "website": "https://github.com/QwenLM/Qwen3.6", + "trust_vector": { + "performance_reliability": { + "overall_score": 91, + "criteria": { + "task_accuracy_code": { + "score": 94, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "SWE-bench Verified 77.2%, SWE-bench Pro 53.5%, Terminal-Bench 2.0 59.3% — surpassing the 397B-A17B Qwen3.5 flagship on agentic coding" + }, + { + "source": "NYU Shanghai RITS analysis", + "url": "https://rits.shanghai.nyu.edu/ai/qwen3-6-27b-a-dense-27b-model-that-beats-a-397b-moe-on-coding", + "date": "2026-04-25", + "value": "Independent analysis confirms the dense 27B edges past the 397B MoE Qwen3.5 on agentic coding benchmarks" + } + ], + "methodology": "Vendor agentic-coding benchmarks corroborated by independent analysis; independent harness runs tend to land below vendor peaks", + "last_verified": "2026-07-09" + }, + "task_accuracy_reasoning": { + "score": 92, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "AIME 2026 94.1%, HMMT Feb 2026 84.3%, GPQA Diamond 87.8% in thinking mode" + }, + { + "source": "Hugging Face Model Card (Qwen3.6-35B-A3B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B", + "date": "2026-04-16", + "value": "AIME 2026 92.7%, GPQA Diamond 86.0% with only 3B active parameters" + } + ], + "methodology": "Mathematical and scientific reasoning benchmarks from official model cards; vendor-reported, pending broader third-party replication", + "last_verified": "2026-07-09" + }, + "task_accuracy_general": { + "score": 89, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "MMLU-Pro 86.2%; vision-language MMMU-Pro 75.8% and RefCOCO avg 92.5% with text, image, and video input" + } + ], + "methodology": "Knowledge and multimodal benchmark review across 201-language coverage; smaller models trail the 397B flagship on knowledge breadth", + "last_verified": "2026-07-09" + }, + "output_consistency": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "Thinking Preservation retains reasoning traces across turns, improving multi-turn agent consistency in community testing" + } + ], + "methodology": "Repeated-prompt and multi-turn agent testing, supplemented by community reports since April 2026", + "last_verified": "2026-07-09" + }, + "latency_p50": { + "value": "27B: ~56 tok/s self-hosted (thinking mode is token-hungry); 35B-A3B: markedly faster with 3B active params", + "confidence": "medium", + "evidence": [ + { + "source": "The AI Rankings - Qwen 3.6 review", + "url": "https://theairankings.com/alibaba/qwen-3-6/", + "date": "2026-07-09", + "value": "27B measured at ~56 tokens/s and described as token-hungry in thinking mode; 3B-active MoE variant targets high-throughput serving" + } + ], + "methodology": "Independent throughput measurements on reference hardware plus architecture analysis", + "last_verified": "2026-07-09" + }, + "latency_p95": { + "value": "Provider-dependent; long thinking traces stretch tail latency unless reasoning is toggled off", + "confidence": "low", + "evidence": [ + { + "source": "The AI Rankings - Qwen 3.6 review", + "url": "https://theairankings.com/alibaba/qwen-3-6/", + "date": "2026-07-09", + "value": "Default-on thinking mode with long reasoning traces; disabling thinking or capping output reduces tail latency" + } + ], + "methodology": "Qualitative assessment; per-provider p95 distributions not yet broadly published", + "last_verified": "2026-07-09" + }, + "context_window": { + "value": "262,144 tokens native (extensible to ~1,010,000 via YaRN)", + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "262,144-token native context, extensible to about 1,010,000 tokens with YaRN scaling" + } + ], + "methodology": "Official specification from model cards", + "last_verified": "2026-07-09" + }, + "uptime": { + "score": 93, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-07-09", + "value": "Hosted on Alibaba Cloud (incl. Singapore region) plus OpenRouter and other third-party hosts; self-hosting adds redundancy" + } + ], + "methodology": "Hosted-platform availability plus redundancy across third-party hosts and self-hosting", + "last_verified": "2026-07-09" + } + }, + "notes": "Remarkable capability density: the 27B dense model beats Alibaba's own 397B-A17B Qwen3.5 flagship on agentic coding (77.2% SWE-bench Verified), and the 35B-A3B delivers near-flagship reasoning with 3B active parameters. Main caveats: vendor benchmarks run above independent evaluations, and default-on thinking mode is token-hungry." + }, + "security": { + "overall_score": 83, + "criteria": { + "prompt_injection_resistance": { + "score": 83, + "confidence": "low", + "evidence": [ + { + "source": "QwenLM GitHub", + "url": "https://github.com/QwenLM/Qwen3.6", + "date": "2026-04-22", + "value": "Safety post-training documented; image and video input widen the injection surface, and dedicated red-team results are not yet published" + } + ], + "methodology": "Testing against OWASP LLM01 patterns including image/video-borne injection; limited third-party data given April 2026 launch", + "last_verified": "2026-07-09" + }, + "jailbreak_resistance": { + "score": 82, + "confidence": "low", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "Safety alignment across the family consistent with Qwen3.5 practice; open weights mean alignment is removable downstream" + } + ], + "methodology": "Adversarial prompt testing; assessment accounts for open-weight modifiability and launch recency", + "last_verified": "2026-07-09" + }, + "data_leakage_prevention": { + "score": 81, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Trust Center", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-04-22", + "value": "Standard data handling on hosted endpoints; self-hosting gives complete data control" + } + ], + "methodology": "Analysis of hosted-platform policies plus the self-hosting option for full data isolation", + "last_verified": "2026-07-09" + }, + "output_safety": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "QwenLM GitHub", + "url": "https://github.com/QwenLM/Qwen3.6", + "date": "2026-04-22", + "value": "Multilingual safety filtering across 201 languages, consistent with the Qwen3.5 family's refusal behavior in community testing" + } + ], + "methodology": "Safety testing across harmful content categories and multiple languages on default weights", + "last_verified": "2026-07-09" + }, + "api_security": { + "score": 86, + "confidence": "high", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-04-22", + "value": "API key authentication, HTTPS, RAM-based access control, and rate limiting on Alibaba Cloud" + } + ], + "methodology": "Review of API security features on the first-party hosted platform", + "last_verified": "2026-07-09" + } + }, + "notes": "Inherits the Qwen family's solid multilingual guardrails, but the April 2026 launch means little independent red-teaming exists yet, and image/video input widens the attack surface. Open weights shift responsibility to deployers who fine-tune." + }, + "privacy_compliance": { + "overall_score": 79, + "criteria": { + "data_residency": { + "value": "China/Singapore (Alibaba Cloud first-party); anywhere via self-hosting or third-party hosts", + "confidence": "high", + "evidence": [ + { + "source": "Alibaba Cloud Regions", + "url": "https://www.alibabacloud.com/en/global-locations", + "date": "2026-04-22", + "value": "First-party hosting on Alibaba Cloud is China-jurisdiction (with a Singapore international region); Apache-2.0 weights allow deployment in any jurisdiction" + } + ], + "methodology": "Review of hosting regions and licensing; China-jurisdiction caveat applies to Alibaba's first-party API, not self-hosted or Western-hosted deployments", + "last_verified": "2026-07-09" + }, + "training_data_optout": { + "score": 86, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio Terms", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-04-22", + "value": "Enterprise tier does not train on customer data; self-hosting removes the concern entirely" + } + ], + "methodology": "Analysis of hosted-platform data usage terms", + "last_verified": "2026-07-09" + }, + "data_retention": { + "value": "Per Alibaba Cloud policy (region-dependent); zero when self-hosted", + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Trust Center", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-04-22", + "value": "Hosted retention follows Alibaba Cloud regional policies; self-hosted deployments retain nothing externally" + } + ], + "methodology": "Review of hosted-platform retention policies; retention is deployment-dependent for open-weight models", + "last_verified": "2026-07-09" + }, + "pii_handling": { + "score": 77, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Documentation", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-04-22", + "value": "Customer responsible for PII redaction; Alibaba Cloud provides surrounding data-governance tooling" + } + ], + "methodology": "Review of data protection capabilities and customer responsibilities", + "last_verified": "2026-07-09" + }, + "compliance_certifications": { + "score": 74, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Trust Center", + "url": "https://www.alibabacloud.com/en/trust-center", + "date": "2026-04-22", + "value": "Alibaba Cloud holds ISO 27001/SOC reports for its infrastructure, but no HIPAA/FedRAMP path for the model service; Western-host deployments inherit those hosts' certifications" + } + ], + "methodology": "Verification of infrastructure certifications versus model-service-level compliance for Western regulated markets", + "last_verified": "2026-07-09" + }, + "zero_data_retention": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Open-weight deployment options", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "No zero-retention guarantee on first-party hosting; self-hosting provides true zero external retention" + } + ], + "methodology": "Review of data handling across first-party API, third-party hosts, and self-hosting", + "last_verified": "2026-07-09" + } + }, + "notes": "Same posture as Qwen3.5: Alibaba's first-party API is China-jurisdiction (Singapore region available), which concerns Western regulated buyers; Apache-2.0 self-hosting or Western third-party hosting fully avoids that. No HIPAA/FedRAMP for the model service. The small footprints (27B dense fits a single 24GB GPU in 4-bit) make compliant self-hosting unusually practical." + }, + "trust_transparency": { + "overall_score": 81, + "criteria": { + "explainability": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "Thinking mode on by default exposes reasoning traces; Thinking Preservation keeps them across turns; fully inspectable when self-hosted" + } + ], + "methodology": "Evaluation of reasoning transparency and trace accessibility", + "last_verified": "2026-07-09" + }, + "hallucination_rate": { + "score": 80, + "confidence": "low", + "evidence": [ + { + "source": "The AI Rankings - Qwen 3.6 review", + "url": "https://theairankings.com/alibaba/qwen-3-6/", + "date": "2026-07-09", + "value": "Independent evaluations land below vendor benchmarks on complex tasks; dedicated factuality studies for Qwen3.6 not yet published" + } + ], + "methodology": "Early factual QA and grounding observations; limited independent data given April 2026 launch", + "last_verified": "2026-07-09" + }, + "bias_fairness": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "Broad multilingual fairness work inherited from the Qwen line; topic-avoidance on China-politically-sensitive subjects persists in default weights" + } + ], + "methodology": "Evaluation on bias benchmarks across languages and politically sensitive topic probes", + "last_verified": "2026-07-09" + }, + "uncertainty_quantification": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "Expresses uncertainty in thinking traces; final-answer calibration adequate but not independently characterized" + } + ], + "methodology": "Qualitative assessment of confidence expression in outputs", + "last_verified": "2026-07-09" + }, + "model_card_quality": { + "score": 90, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-35B-A3B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B", + "date": "2026-04-16", + "value": "Detailed cards: hybrid Gated DeltaNet + Gated Attention architecture, MoE layout (256 experts, 8 routed + 1 shared), benchmarks, output-length guidance, and deployment recipes" + } + ], + "methodology": "Review of model card and technical documentation completeness", + "last_verified": "2026-07-09" + }, + "training_data_transparency": { + "score": 75, + "confidence": "medium", + "evidence": [ + { + "source": "QwenLM GitHub", + "url": "https://github.com/QwenLM/Qwen3.6", + "date": "2026-04-22", + "value": "Training methodology described at a high level; detailed data composition not disclosed, consistent with prior Qwen releases" + } + ], + "methodology": "Review of public disclosures about training data", + "last_verified": "2026-07-09" + }, + "guardrails": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "QwenLM GitHub", + "url": "https://github.com/QwenLM/Qwen3.6", + "date": "2026-04-22", + "value": "Multilingual safety alignment in released weights; removable by downstream fine-tuning" + } + ], + "methodology": "Analysis of built-in safety mechanisms in default weights", + "last_verified": "2026-07-09" + } + }, + "notes": "Strong open documentation and unusually inspectable reasoning via default-on thinking with Thinking Preservation. Typical Qwen gaps remain: limited training-data detail, topic-avoidance on politically sensitive subjects, and — given the recent launch — sparse independent factuality data." + }, + "operational_excellence": { + "overall_score": 86, + "criteria": { + "api_design_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-04-22", + "value": "OpenAI-compatible API with function calling, multimodal inputs, and thinking-mode toggles" + } + ], + "methodology": "Review of API design, consistency, and feature completeness", + "last_verified": "2026-07-09" + }, + "sdk_quality": { + "score": 88, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "Day-one support in vLLM, SGLang, KTransformers, and Hugging Face Transformers; actively maintained official repos" + } + ], + "methodology": "Review of SDK and inference-framework support", + "last_verified": "2026-07-09" + }, + "versioning_policy": { + "score": 82, + "confidence": "medium", + "evidence": [ + { + "source": "The AI Rankings - Qwen 3.6 review", + "url": "https://theairankings.com/alibaba/qwen-3-6/", + "date": "2026-07-09", + "value": "Staged open releases (35B-A3B 2026-04-16, 27B 2026-04-22) two months after Qwen3.5; open weights remain permanently available, softening the fast cadence" + }, + { + "source": "innFactory - Qwen model overview", + "url": "https://innfactory.ai/en/ai-models/qwen/", + "date": "2026-07-09", + "value": "Family frontier has moved to proprietary API-only Qwen3.7-Max (May 2026) and Qwen3.7-Plus (Jun 2026); Qwen3.6 is the current open-weight generation" + } + ], + "methodology": "Review of release cadence and weight-availability guarantees", + "last_verified": "2026-07-09" + }, + "monitoring_observability": { + "score": 84, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Model Studio", + "url": "https://www.alibabacloud.com/en/product/modelstudio", + "date": "2026-04-22", + "value": "Usage dashboards and logging on Alibaba Cloud; full observability when self-hosting" + } + ], + "methodology": "Review of monitoring tools across deployment options", + "last_verified": "2026-07-09" + }, + "support_quality": { + "score": 78, + "confidence": "medium", + "evidence": [ + { + "source": "Alibaba Cloud Support", + "url": "https://www.alibabacloud.com/en/contact-sales", + "date": "2026-04-22", + "value": "Alibaba Cloud offers paid enterprise support tiers; Western-market support depth lags US hyperscalers; strong community channels" + } + ], + "methodology": "Assessment of support tiers, documentation, and community responsiveness", + "last_verified": "2026-07-09" + }, + "ecosystem_maturity": { + "score": 87, + "confidence": "medium", + "evidence": [ + { + "source": "Hugging Face Qwen Organization", + "url": "https://huggingface.co/Qwen", + "date": "2026-07-09", + "value": "Inherits the largest open-model ecosystem by derivative count; Qwen3.6-specific quantizations and fine-tunes accumulating since April 2026 but younger than the Qwen3.5 ladder" + } + ], + "methodology": "Analysis of derivative models, third-party hosting, and tooling integrations; scored slightly conservative given the ~3-month age", + "last_verified": "2026-07-09" + }, + "license_terms": { + "score": 98, + "confidence": "high", + "evidence": [ + { + "source": "Hugging Face Model Card (Qwen3.6-27B)", + "url": "https://huggingface.co/Qwen/Qwen3.6-27B", + "date": "2026-04-22", + "value": "Apache 2.0 across both released models: unrestricted commercial use with explicit patent grant" + } + ], + "methodology": "Review of licensing terms and restrictions", + "last_verified": "2026-07-09" + } + }, + "notes": "Rides the mature Qwen ecosystem: Apache 2.0 with patent grant, day-one vLLM/SGLang/KTransformers/transformers support, and hosted options from Alibaba Cloud ($0.60/$3.60 per 1M for 27B) to cheaper third-party hosts. Only two sizes released so far — the family ladder is thinner than Qwen3.5's 0.8B-397B range — and the newest family frontier (Qwen3.7-Max/Plus) is API-only." + } + }, + "use_case_ratings": { + "code-generation": { + "overall": 93, + "notes": "77.2% SWE-bench Verified from a 27B dense model that self-hosts on a single 24GB GPU in 4-bit — exceptional agentic-coding economics; beats the 397B-A17B Qwen3.5 flagship.", + "alternatives": ["kimi-k2-6", "gpt-5-3-codex"] + }, + "customer-support": { + "overall": 87, + "notes": "201-language coverage with a 3B-active MoE variant that serves high-volume tiers very cheaply; thinking mode should be toggled off for latency.", + "alternatives": ["qwen3-5", "claude-haiku-4-5"] + }, + "content-creation": { + "overall": 85, + "notes": "Strong multilingual content with image and video understanding for visually grounded writing.", + "alternatives": ["claude-opus-4-6", "qwen3-5"] + }, + "data-analysis": { + "overall": 88, + "notes": "Native image/video input handles charts and documents; 262K context (~1M via YaRN) covers large datasets at small-model cost.", + "alternatives": ["gemini-3-pro", "deepseek-v4"] + }, + "research-assistant": { + "overall": 88, + "notes": "Multimodal document understanding, long context, and preserved reasoning traces suit iterative research workflows.", + "alternatives": ["deepseek-v4", "claude-opus-4-6"] + }, + "legal-compliance": { + "overall": 75, + "notes": "First-party hosting is China-jurisdiction; viable for regulated legal work only via self-hosting or certified Western hosts.", + "alternatives": ["claude-opus-4-6"] + }, + "healthcare": { + "overall": 73, + "notes": "No HIPAA path on first-party hosting; the small footprint makes self-hosted deployment in compliant infrastructure practical.", + "alternatives": ["claude-opus-4-6"] + }, + "financial-analysis": { + "overall": 86, + "notes": "Strong quantitative reasoning (AIME 2026 94.1%) with chart/table understanding; data-residency planning required for regulated workloads.", + "alternatives": ["deepseek-v4", "gpt-5-5"] + }, + "education": { + "overall": 90, + "notes": "201 languages, multimodal input, visible reasoning traces, and single-GPU deployability make it excellent for global and budget education deployments.", + "alternatives": ["qwen3-5", "deepseek-v3-2"] + }, + "creative-writing": { + "overall": 83, + "notes": "Capable multilingual creative output with visual grounding; prose distinctiveness behind dedicated creative leaders.", + "alternatives": ["claude-opus-4-6", "deepseek-v4"] + } + }, + "strengths": [ + "27B dense model beats the 397B-A17B Qwen3.5 flagship on agentic coding: 77.2% SWE-bench Verified, 59.3% Terminal-Bench 2.0", + "Exceptional efficiency: 27B runs in ~18GB VRAM (4-bit) on a single 24GB GPU; 35B-A3B activates only 3B parameters", + "Apache 2.0 with patent grant across both released models", + "Multimodal input (text, image, video) across 201 languages and dialects", + "262K native context, extensible to ~1M tokens via YaRN", + "Thinking Preservation keeps reasoning traces across turns, improving iterative agent workflows", + "Day-one vLLM/SGLang/KTransformers/transformers support within the largest open-model ecosystem" + ], + "limitations": [ + "First-party Alibaba Cloud hosting is China-jurisdiction (Singapore region available); no HIPAA/FedRAMP path for the model service — self-hosting or Western hosts avoid this", + "Vendor benchmarks notably exceed independent evaluations; still below closed frontier models on the most complex tasks", + "Token-hungry and relatively slow in default thinking mode (~56 tok/s for the 27B self-hosted)", + "Only two sizes released (27B dense, 35B-A3B) versus Qwen3.5's full 0.8B-397B ladder; family frontier Qwen3.7-Max/Plus is API-only", + "Topic-avoidance on politically sensitive subjects in default weights", + "Training-data composition disclosed only at a high level", + "Launched April 2026: limited independent red-teaming and factuality studies so far" + ], + "best_for": [ + "Agentic coding assistants with frontier-adjacent quality on single-GPU budgets", + "Cost-efficient high-volume serving via the 3B-active 35B-A3B MoE", + "Global multilingual products needing 201-language coverage under an open license", + "Multimodal agents processing documents, images, and video natively", + "Compliant self-hosted deployments where the small footprint keeps infrastructure practical" + ], + "not_recommended_for": [ + "Regulated Western workloads via Alibaba's first-party API", + "Latency-critical applications running default thinking mode", + "Audio input or any media generation applications", + "Frontier-difficulty tasks where closed models still lead", + "Teams needing a broad open size ladder today (only 27B and 35B-A3B released)" + ], + "metadata": { + "pricing": { + "input": "Free weights (Apache 2.0); hosted 27B $0.60 per 1M tokens on Alibaba Cloud (Singapore), 35B-A3B ~$0.15; OpenRouter from ~$0.14-$0.29", + "output": "Hosted 27B $3.60 per 1M tokens on Alibaba Cloud; OpenRouter ~$1.00 (35B-A3B) to ~$3.17 (27B)", + "notes": "Chinese Mainland endpoint is 60-70% cheaper than Singapore; batch invocation takes 50% off. Self-hosting is infrastructure-cost-only — the 27B fits a single 24GB GPU in 4-bit. Thinking-mode verbosity inflates output-token spend.", + "last_verified": "2026-07-09" + }, + "context_window": 262144, + "max_output": 81920, + "languages": [ + "English", + "Chinese", + "Japanese", + "Korean", + "Spanish", + "French", + "German", + "Portuguese", + "Russian", + "Arabic", + "Hindi", + "Indonesian", + "Vietnamese", + "Thai", + "and 187 more (201 total)" + ], + "modalities": ["text", "image (input)", "video (input)"], + "api_endpoint": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions", + "open_source": true, + "architecture": "Hybrid attention alternating Gated DeltaNet and Gated Attention layers; 27B dense (64 layers) and 35B-A3B MoE (256 experts, 8 routed + 1 shared); default-on thinking mode with Thinking Preservation", + "parameters": "27B dense; 35B total / 3B active (MoE)", + "knowledge_cutoff": "Not officially disclosed" + }, + "related_entities": ["qwen3-5", "deepseek-v4", "kimi-k2-6", "glm-5", "minimax-m2"], + "tags": [ + "open-source", + "apache-2-0", + "multimodal", + "multilingual", + "agentic", + "coding", + "long-context", + "chinese-provider", + "efficient" + ] +} From 062efe5c1d053f8092b545d24677e542f758c22c Mon Sep 17 00:00:00 2001 From: JBAhire Date: Thu, 9 Jul 2026 22:53:12 -0700 Subject: [PATCH 3/7] feat: auto-generated data index + client summary bundle Replace the hand-maintained 158-import lib/data.ts with codegen: scripts/generate-data-index.ts scans data/ and emits lib/data-index.ts (full entities, server-side) and lib/data-summaries.ts (lightweight EntitySummary literals for client views). List/compare pages now ship ~158KB of summaries instead of ~4.4MB of full JSON: homepage first-load JS drops 665kB -> 138kB (-79%). Also: draft gate (_-prefixed files or "draft": true stay unpublished), identifier-collision detection, predev/prebuild/pretype-check/prevalidate hooks, CI drift check for the generated index, memoized EntityCard, and cached overall scores for sorting. Co-Authored-By: Claude Fable 5 --- .github/workflows/validate.yml | 5 + lib/client-data.ts | 72 + lib/data-index.ts | 411 +++ lib/data-summaries.ts | 5279 ++++++++++++++++++++++++++++++++ lib/data.ts | 463 +-- lib/summary-types.ts | 34 + lib/utils.ts | 3 + package-lock.json | 15 + package.json | 7 +- scripts/generate-data-index.ts | 190 ++ 10 files changed, 6041 insertions(+), 438 deletions(-) create mode 100644 lib/client-data.ts create mode 100644 lib/data-index.ts create mode 100644 lib/data-summaries.ts create mode 100644 lib/summary-types.ts create mode 100644 scripts/generate-data-index.ts diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml index 25be197..176838a 100644 --- a/.github/workflows/validate.yml +++ b/.github/workflows/validate.yml @@ -35,6 +35,11 @@ jobs: - name: Run validation run: npm run validate + - name: Check data index is in sync + run: | + npm run generate:data-index + git diff --exit-code lib/data-index.ts || (echo "lib/data-index.ts is stale — run 'npm run generate:data-index' and commit it" && exit 1) + - name: Type check run: npm run type-check diff --git a/lib/client-data.ts b/lib/client-data.ts new file mode 100644 index 0000000..8ea4a2b --- /dev/null +++ b/lib/client-data.ts @@ -0,0 +1,72 @@ +/** + * Client-side data utilities for TrustVector list/compare views. + * + * Works ONLY on lightweight EntitySummary objects from the auto-generated + * lib/data-summaries.ts (~5% of the full dataset), so 'use client' pages + * never bundle the full evaluation JSON. Server components that need full + * entities (detail pages, sitemap) use lib/data.ts instead. + * + * Must not import lib/data.ts or lib/data-index.ts — doing so would pull + * the entire dataset back into the client bundle. + */ + +import type { EntitySummary } from './summary-types'; +import { ALL_SUMMARIES } from './data-summaries'; + +/** + * Get all entity summaries + */ +export function getAllSummaries(): EntitySummary[] { + return ALL_SUMMARIES; +} + +/** + * Search summaries by query (matches name, provider, description, tags) + */ +export function searchSummaries(query: string): EntitySummary[] { + const lowerQuery = query.toLowerCase(); + return ALL_SUMMARIES.filter((summary) => { + return ( + summary.name.toLowerCase().includes(lowerQuery) || + summary.provider.toLowerCase().includes(lowerQuery) || + summary.description.toLowerCase().includes(lowerQuery) || + summary.tags.some((tag) => tag.toLowerCase().includes(lowerQuery)) + ); + }); +} + +/** + * Sort summaries — mirrors lib/data.ts sortEntities option-for-option, + * using the precomputed summary.overall_score for score sorts. + */ +export type SortOption = + | 'name-asc' + | 'name-desc' + | 'overall-score-desc' + | 'overall-score-asc' + | 'date-desc' + | 'date-asc'; + +export function sortSummaries( + summaries: EntitySummary[], + sortBy: SortOption +): EntitySummary[] { + const sorted = [...summaries]; + + switch (sortBy) { + case 'name-asc': + return sorted.sort((a, b) => a.name.localeCompare(b.name)); + case 'name-desc': + return sorted.sort((a, b) => b.name.localeCompare(a.name)); + case 'overall-score-desc': + return sorted.sort((a, b) => b.overall_score - a.overall_score); + case 'overall-score-asc': + return sorted.sort((a, b) => a.overall_score - b.overall_score); + case 'date-desc': + return sorted.sort((a, b) => b.last_evaluated.localeCompare(a.last_evaluated)); + case 'date-asc': + return sorted.sort((a, b) => a.last_evaluated.localeCompare(b.last_evaluated)); + default: + return sorted; + } +} diff --git a/lib/data-index.ts b/lib/data-index.ts new file mode 100644 index 0000000..75bf0bb --- /dev/null +++ b/lib/data-index.ts @@ -0,0 +1,411 @@ +/** + * AUTO-GENERATED by scripts/generate-data-index.ts — do not edit by hand. + * Regenerate with: npm run generate:data-index + * + * SERVER-SIDE ONLY: bundles the full evaluation dataset. Import via + * lib/data.ts from server components only — never from 'use client' code. + * Client list/compare views use lib/data-summaries.ts instead. + * + * 196 entities: 68 models, 67 agents, 61 MCP servers. + */ + +import type { TrustVectorEntity } from '@/framework/schema/types'; + +import models_claudeFable5 from '@/data/models/claude-fable-5.json'; +import models_claudeHaiku45 from '@/data/models/claude-haiku-4-5.json'; +import models_claudeOpus41 from '@/data/models/claude-opus-4-1.json'; +import models_claudeOpus45 from '@/data/models/claude-opus-4-5.json'; +import models_claudeOpus46 from '@/data/models/claude-opus-4-6.json'; +import models_claudeOpus47 from '@/data/models/claude-opus-4-7.json'; +import models_claudeOpus48 from '@/data/models/claude-opus-4-8.json'; +import models_claudeOpus4 from '@/data/models/claude-opus-4.json'; +import models_claudeSonnet45 from '@/data/models/claude-sonnet-4-5.json'; +import models_claudeSonnet46 from '@/data/models/claude-sonnet-4-6.json'; +import models_claudeSonnet4 from '@/data/models/claude-sonnet-4.json'; +import models_claudeSonnet5 from '@/data/models/claude-sonnet-5.json'; +import models_commandAPlus from '@/data/models/command-a-plus.json'; +import models_deepseekR1 from '@/data/models/deepseek-r1.json'; +import models_deepseekV30324 from '@/data/models/deepseek-v3-0324.json'; +import models_deepseekV32 from '@/data/models/deepseek-v3-2.json'; +import models_deepseekV4 from '@/data/models/deepseek-v4.json'; +import models_gemini20Flash from '@/data/models/gemini-2-0-flash.json'; +import models_gemini25Pro from '@/data/models/gemini-2-5-pro.json'; +import models_gemini31Pro from '@/data/models/gemini-3-1-pro.json'; +import models_gemini35Flash from '@/data/models/gemini-3-5-flash.json'; +import models_gemini3Flash from '@/data/models/gemini-3-flash.json'; +import models_gemini3Pro from '@/data/models/gemini-3-pro.json'; +import models_gemma327b from '@/data/models/gemma-3-27b.json'; +import models_gemma4 from '@/data/models/gemma-4.json'; +import models_glm52 from '@/data/models/glm-5-2.json'; +import models_glm5 from '@/data/models/glm-5.json'; +import models_gpt41Mini from '@/data/models/gpt-4-1-mini.json'; +import models_gpt41Nano from '@/data/models/gpt-4-1-nano.json'; +import models_gpt41 from '@/data/models/gpt-4-1.json'; +import models_gpt4oMini from '@/data/models/gpt-4o-mini.json'; +import models_gpt4o from '@/data/models/gpt-4o.json'; +import models_gpt51 from '@/data/models/gpt-5-1.json'; +import models_gpt52Codex from '@/data/models/gpt-5-2-codex.json'; +import models_gpt52 from '@/data/models/gpt-5-2.json'; +import models_gpt53Codex from '@/data/models/gpt-5-3-codex.json'; +import models_gpt54 from '@/data/models/gpt-5-4.json'; +import models_gpt55 from '@/data/models/gpt-5-5.json'; +import models_gpt56 from '@/data/models/gpt-5-6.json'; +import models_gpt5 from '@/data/models/gpt-5.json'; +import models_gptOss120b from '@/data/models/gpt-oss-120b.json'; +import models_gptOss20b from '@/data/models/gpt-oss-20b.json'; +import models_grok3Beta from '@/data/models/grok-3-beta.json'; +import models_grok41 from '@/data/models/grok-4-1.json'; +import models_grok43 from '@/data/models/grok-4-3.json'; +import models_grok45 from '@/data/models/grok-4-5.json'; +import models_kimiK26 from '@/data/models/kimi-k2-6.json'; +import models_kimiK27Code from '@/data/models/kimi-k2-7-code.json'; +import models_llama31405b from '@/data/models/llama-3-1-405b.json'; +import models_llama3370b from '@/data/models/llama-3-3-70b.json'; +import models_llama4Behemoth from '@/data/models/llama-4-behemoth.json'; +import models_llama4Maverick from '@/data/models/llama-4-maverick.json'; +import models_llama4Scout from '@/data/models/llama-4-scout.json'; +import models_minimaxM2 from '@/data/models/minimax-m2.json'; +import models_minimaxM3 from '@/data/models/minimax-m3.json'; +import models_mistralLarge3 from '@/data/models/mistral-large-3.json'; +import models_nemotron3Ultra from '@/data/models/nemotron-3-ultra.json'; +import models_nemotronUltra253b from '@/data/models/nemotron-ultra-253b.json'; +import models_nova2Lite from '@/data/models/nova-2-lite.json'; +import models_novaPro from '@/data/models/nova-pro.json'; +import models_openaiO1Mini from '@/data/models/openai-o1-mini.json'; +import models_openaiO1 from '@/data/models/openai-o1.json'; +import models_openaiO3Mini from '@/data/models/openai-o3-mini.json'; +import models_openaiO3 from '@/data/models/openai-o3.json'; +import models_openaiO4Mini from '@/data/models/openai-o4-mini.json'; +import models_qwen25Vl32b from '@/data/models/qwen2-5-vl-32b.json'; +import models_qwen35 from '@/data/models/qwen3-5.json'; +import models_qwen36 from '@/data/models/qwen3-6.json'; +import agents_activepieces from '@/data/agents/activepieces.json'; +import agents_adala from '@/data/agents/adala.json'; +import agents_agentgpt from '@/data/agents/agentgpt.json'; +import agents_amazonBedrockAgents from '@/data/agents/amazon-bedrock-agents.json'; +import agents_amazonKiro from '@/data/agents/amazon-kiro.json'; +import agents_amazonLex from '@/data/agents/amazon-lex.json'; +import agents_autogen from '@/data/agents/autogen.json'; +import agents_autogpt from '@/data/agents/autogpt.json'; +import agents_azureBotService from '@/data/agents/azure-bot-service.json'; +import agents_babyagi from '@/data/agents/babyagi.json'; +import agents_bytedanceTrae from '@/data/agents/bytedance-trae.json'; +import agents_chatgptAgent from '@/data/agents/chatgpt-agent.json'; +import agents_claudeAgentSdk from '@/data/agents/claude-agent-sdk.json'; +import agents_claudeCode from '@/data/agents/claude-code.json'; +import agents_claudeCowork from '@/data/agents/claude-cowork.json'; +import agents_cline from '@/data/agents/cline.json'; +import agents_crewai from '@/data/agents/crewai.json'; +import agents_cursorAgent from '@/data/agents/cursor-agent.json'; +import agents_devin from '@/data/agents/devin.json'; +import agents_dify from '@/data/agents/dify.json'; +import agents_e2bAgents from '@/data/agents/e2b-agents.json'; +import agents_factoryDroids from '@/data/agents/factory-droids.json'; +import agents_flowise from '@/data/agents/flowise.json'; +import agents_geminiCli from '@/data/agents/gemini-cli.json'; +import agents_githubCopilotCodingAgent from '@/data/agents/github-copilot-coding-agent.json'; +import agents_gleanAi from '@/data/agents/glean-ai.json'; +import agents_googleAdk from '@/data/agents/google-adk.json'; +import agents_googleAgentBuilder from '@/data/agents/google-agent-builder.json'; +import agents_googleAntigravity from '@/data/agents/google-antigravity.json'; +import agents_googleDialogflow from '@/data/agents/google-dialogflow.json'; +import agents_googleJules from '@/data/agents/google-jules.json'; +import agents_goose from '@/data/agents/goose.json'; +import agents_haystack from '@/data/agents/haystack.json'; +import agents_ibmWatsonAssistant from '@/data/agents/ibm-watson-assistant.json'; +import agents_jetbrainsJunie from '@/data/agents/jetbrains-junie.json'; +import agents_koreAi from '@/data/agents/kore-ai.json'; +import agents_langflow from '@/data/agents/langflow.json'; +import agents_langgraphAgent from '@/data/agents/langgraph-agent.json'; +import agents_llamaindexAgent from '@/data/agents/llamaindex-agent.json'; +import agents_lovable from '@/data/agents/lovable.json'; +import agents_makeAi from '@/data/agents/make-ai.json'; +import agents_manus from '@/data/agents/manus.json'; +import agents_mastra from '@/data/agents/mastra.json'; +import agents_memgpt from '@/data/agents/memgpt.json'; +import agents_microsoftAgentFramework from '@/data/agents/microsoft-agent-framework.json'; +import agents_microsoftScout from '@/data/agents/microsoft-scout.json'; +import agents_n8nAiAgent from '@/data/agents/n8n-ai-agent.json'; +import agents_openaiAgentsSdk from '@/data/agents/openai-agents-sdk.json'; +import agents_openaiAssistantsApi from '@/data/agents/openai-assistants-api.json'; +import agents_openaiCodex from '@/data/agents/openai-codex.json'; +import agents_openclaw from '@/data/agents/openclaw.json'; +import agents_opencode from '@/data/agents/opencode.json'; +import agents_perplexityComet from '@/data/agents/perplexity-comet.json'; +import agents_poke from '@/data/agents/poke.json'; +import agents_pydanticAi from '@/data/agents/pydantic-ai.json'; +import agents_rasa from '@/data/agents/rasa.json'; +import agents_relevanceAi from '@/data/agents/relevance-ai.json'; +import agents_replitAgent from '@/data/agents/replit-agent.json'; +import agents_salesforceEinsteinBots from '@/data/agents/salesforce-einstein-bots.json'; +import agents_semanticKernelAgent from '@/data/agents/semantic-kernel-agent.json'; +import agents_sierraAi from '@/data/agents/sierra-ai.json'; +import agents_smolagents from '@/data/agents/smolagents.json'; +import agents_strandsAgents from '@/data/agents/strands-agents.json'; +import agents_superagi from '@/data/agents/superagi.json'; +import agents_swarm from '@/data/agents/swarm.json'; +import agents_warp from '@/data/agents/warp.json'; +import agents_zapierAi from '@/data/agents/zapier-ai.json'; +import mcps_mcpServerApify from '@/data/mcps/mcp-server-apify.json'; +import mcps_mcpServerAsana from '@/data/mcps/mcp-server-asana.json'; +import mcps_mcpServerAtlassian from '@/data/mcps/mcp-server-atlassian.json'; +import mcps_mcpServerAws from '@/data/mcps/mcp-server-aws.json'; +import mcps_mcpServerAzure from '@/data/mcps/mcp-server-azure.json'; +import mcps_mcpServerBox from '@/data/mcps/mcp-server-box.json'; +import mcps_mcpServerBraveSearch from '@/data/mcps/mcp-server-brave-search.json'; +import mcps_mcpServerBrowserbase from '@/data/mcps/mcp-server-browserbase.json'; +import mcps_mcpServerCalendar from '@/data/mcps/mcp-server-calendar.json'; +import mcps_mcpServerCanva from '@/data/mcps/mcp-server-canva.json'; +import mcps_mcpServerChromeDevtools from '@/data/mcps/mcp-server-chrome-devtools.json'; +import mcps_mcpServerClickhouse from '@/data/mcps/mcp-server-clickhouse.json'; +import mcps_mcpServerCloudflare from '@/data/mcps/mcp-server-cloudflare.json'; +import mcps_mcpServerContext7 from '@/data/mcps/mcp-server-context7.json'; +import mcps_mcpServerDatabricks from '@/data/mcps/mcp-server-databricks.json'; +import mcps_mcpServerDatadog from '@/data/mcps/mcp-server-datadog.json'; +import mcps_mcpServerDocker from '@/data/mcps/mcp-server-docker.json'; +import mcps_mcpServerElasticsearch from '@/data/mcps/mcp-server-elasticsearch.json'; +import mcps_mcpServerEverything from '@/data/mcps/mcp-server-everything.json'; +import mcps_mcpServerExa from '@/data/mcps/mcp-server-exa.json'; +import mcps_mcpServerFetch from '@/data/mcps/mcp-server-fetch.json'; +import mcps_mcpServerFigma from '@/data/mcps/mcp-server-figma.json'; +import mcps_mcpServerFilesystem from '@/data/mcps/mcp-server-filesystem.json'; +import mcps_mcpServerFirecrawl from '@/data/mcps/mcp-server-firecrawl.json'; +import mcps_mcpServerGit from '@/data/mcps/mcp-server-git.json'; +import mcps_mcpServerGithub from '@/data/mcps/mcp-server-github.json'; +import mcps_mcpServerGitlab from '@/data/mcps/mcp-server-gitlab.json'; +import mcps_mcpServerGmail from '@/data/mcps/mcp-server-gmail.json'; +import mcps_mcpServerGoogleDrive from '@/data/mcps/mcp-server-google-drive.json'; +import mcps_mcpServerGrafana from '@/data/mcps/mcp-server-grafana.json'; +import mcps_mcpServerHubspot from '@/data/mcps/mcp-server-hubspot.json'; +import mcps_mcpServerHuggingFace from '@/data/mcps/mcp-server-hugging-face.json'; +import mcps_mcpServerKubernetes from '@/data/mcps/mcp-server-kubernetes.json'; +import mcps_mcpServerLinear from '@/data/mcps/mcp-server-linear.json'; +import mcps_mcpServerMemory from '@/data/mcps/mcp-server-memory.json'; +import mcps_mcpServerMongodb from '@/data/mcps/mcp-server-mongodb.json'; +import mcps_mcpServerNeo4j from '@/data/mcps/mcp-server-neo4j.json'; +import mcps_mcpServerNeon from '@/data/mcps/mcp-server-neon.json'; +import mcps_mcpServerNotion from '@/data/mcps/mcp-server-notion.json'; +import mcps_mcpServerPaypal from '@/data/mcps/mcp-server-paypal.json'; +import mcps_mcpServerPerplexity from '@/data/mcps/mcp-server-perplexity.json'; +import mcps_mcpServerPlaywright from '@/data/mcps/mcp-server-playwright.json'; +import mcps_mcpServerPostgres from '@/data/mcps/mcp-server-postgres.json'; +import mcps_mcpServerPuppeteer from '@/data/mcps/mcp-server-puppeteer.json'; +import mcps_mcpServerRedis from '@/data/mcps/mcp-server-redis.json'; +import mcps_mcpServerS3 from '@/data/mcps/mcp-server-s3.json'; +import mcps_mcpServerSalesforce from '@/data/mcps/mcp-server-salesforce.json'; +import mcps_mcpServerSentry from '@/data/mcps/mcp-server-sentry.json'; +import mcps_mcpServerSequentialThinking from '@/data/mcps/mcp-server-sequential-thinking.json'; +import mcps_mcpServerSerena from '@/data/mcps/mcp-server-serena.json'; +import mcps_mcpServerShadcn from '@/data/mcps/mcp-server-shadcn.json'; +import mcps_mcpServerShopify from '@/data/mcps/mcp-server-shopify.json'; +import mcps_mcpServerSlack from '@/data/mcps/mcp-server-slack.json'; +import mcps_mcpServerSnowflake from '@/data/mcps/mcp-server-snowflake.json'; +import mcps_mcpServerSqlite from '@/data/mcps/mcp-server-sqlite.json'; +import mcps_mcpServerStripe from '@/data/mcps/mcp-server-stripe.json'; +import mcps_mcpServerSupabase from '@/data/mcps/mcp-server-supabase.json'; +import mcps_mcpServerTavily from '@/data/mcps/mcp-server-tavily.json'; +import mcps_mcpServerTime from '@/data/mcps/mcp-server-time.json'; +import mcps_mcpServerVercel from '@/data/mcps/mcp-server-vercel.json'; +import mcps_mcpServerZapier from '@/data/mcps/mcp-server-zapier.json'; + +export const ALL_ENTITIES: TrustVectorEntity[] = [ + // models (68) + models_claudeFable5, + models_claudeHaiku45, + models_claudeOpus41, + models_claudeOpus45, + models_claudeOpus46, + models_claudeOpus47, + models_claudeOpus48, + models_claudeOpus4, + models_claudeSonnet45, + models_claudeSonnet46, + models_claudeSonnet4, + models_claudeSonnet5, + models_commandAPlus, + models_deepseekR1, + models_deepseekV30324, + models_deepseekV32, + models_deepseekV4, + models_gemini20Flash, + models_gemini25Pro, + models_gemini31Pro, + models_gemini35Flash, + models_gemini3Flash, + models_gemini3Pro, + models_gemma327b, + models_gemma4, + models_glm52, + models_glm5, + models_gpt41Mini, + models_gpt41Nano, + models_gpt41, + models_gpt4oMini, + models_gpt4o, + models_gpt51, + models_gpt52Codex, + models_gpt52, + models_gpt53Codex, + models_gpt54, + models_gpt55, + models_gpt56, + models_gpt5, + models_gptOss120b, + models_gptOss20b, + models_grok3Beta, + models_grok41, + models_grok43, + models_grok45, + models_kimiK26, + models_kimiK27Code, + models_llama31405b, + models_llama3370b, + models_llama4Behemoth, + models_llama4Maverick, + models_llama4Scout, + models_minimaxM2, + models_minimaxM3, + models_mistralLarge3, + models_nemotron3Ultra, + models_nemotronUltra253b, + models_nova2Lite, + models_novaPro, + models_openaiO1Mini, + models_openaiO1, + models_openaiO3Mini, + models_openaiO3, + models_openaiO4Mini, + models_qwen25Vl32b, + models_qwen35, + models_qwen36, + // agents (67) + agents_activepieces, + agents_adala, + agents_agentgpt, + agents_amazonBedrockAgents, + agents_amazonKiro, + agents_amazonLex, + agents_autogen, + agents_autogpt, + agents_azureBotService, + agents_babyagi, + agents_bytedanceTrae, + agents_chatgptAgent, + agents_claudeAgentSdk, + agents_claudeCode, + agents_claudeCowork, + agents_cline, + agents_crewai, + agents_cursorAgent, + agents_devin, + agents_dify, + agents_e2bAgents, + agents_factoryDroids, + agents_flowise, + agents_geminiCli, + agents_githubCopilotCodingAgent, + agents_gleanAi, + agents_googleAdk, + agents_googleAgentBuilder, + agents_googleAntigravity, + agents_googleDialogflow, + agents_googleJules, + agents_goose, + agents_haystack, + agents_ibmWatsonAssistant, + agents_jetbrainsJunie, + agents_koreAi, + agents_langflow, + agents_langgraphAgent, + agents_llamaindexAgent, + agents_lovable, + agents_makeAi, + agents_manus, + agents_mastra, + agents_memgpt, + agents_microsoftAgentFramework, + agents_microsoftScout, + agents_n8nAiAgent, + agents_openaiAgentsSdk, + agents_openaiAssistantsApi, + agents_openaiCodex, + agents_openclaw, + agents_opencode, + agents_perplexityComet, + agents_poke, + agents_pydanticAi, + agents_rasa, + agents_relevanceAi, + agents_replitAgent, + agents_salesforceEinsteinBots, + agents_semanticKernelAgent, + agents_sierraAi, + agents_smolagents, + agents_strandsAgents, + agents_superagi, + agents_swarm, + agents_warp, + agents_zapierAi, + // mcps (61) + mcps_mcpServerApify, + mcps_mcpServerAsana, + mcps_mcpServerAtlassian, + mcps_mcpServerAws, + mcps_mcpServerAzure, + mcps_mcpServerBox, + mcps_mcpServerBraveSearch, + mcps_mcpServerBrowserbase, + mcps_mcpServerCalendar, + mcps_mcpServerCanva, + mcps_mcpServerChromeDevtools, + mcps_mcpServerClickhouse, + mcps_mcpServerCloudflare, + mcps_mcpServerContext7, + mcps_mcpServerDatabricks, + mcps_mcpServerDatadog, + mcps_mcpServerDocker, + mcps_mcpServerElasticsearch, + mcps_mcpServerEverything, + mcps_mcpServerExa, + mcps_mcpServerFetch, + mcps_mcpServerFigma, + mcps_mcpServerFilesystem, + mcps_mcpServerFirecrawl, + mcps_mcpServerGit, + mcps_mcpServerGithub, + mcps_mcpServerGitlab, + mcps_mcpServerGmail, + mcps_mcpServerGoogleDrive, + mcps_mcpServerGrafana, + mcps_mcpServerHubspot, + mcps_mcpServerHuggingFace, + mcps_mcpServerKubernetes, + mcps_mcpServerLinear, + mcps_mcpServerMemory, + mcps_mcpServerMongodb, + mcps_mcpServerNeo4j, + mcps_mcpServerNeon, + mcps_mcpServerNotion, + mcps_mcpServerPaypal, + mcps_mcpServerPerplexity, + mcps_mcpServerPlaywright, + mcps_mcpServerPostgres, + mcps_mcpServerPuppeteer, + mcps_mcpServerRedis, + mcps_mcpServerS3, + mcps_mcpServerSalesforce, + mcps_mcpServerSentry, + mcps_mcpServerSequentialThinking, + mcps_mcpServerSerena, + mcps_mcpServerShadcn, + mcps_mcpServerShopify, + mcps_mcpServerSlack, + mcps_mcpServerSnowflake, + mcps_mcpServerSqlite, + mcps_mcpServerStripe, + mcps_mcpServerSupabase, + mcps_mcpServerTavily, + mcps_mcpServerTime, + mcps_mcpServerVercel, + mcps_mcpServerZapier, +] as TrustVectorEntity[]; diff --git a/lib/data-summaries.ts b/lib/data-summaries.ts new file mode 100644 index 0000000..5744807 --- /dev/null +++ b/lib/data-summaries.ts @@ -0,0 +1,5279 @@ +/** + * AUTO-GENERATED by scripts/generate-data-index.ts — do not edit by hand. + * Regenerate with: npm run generate:data-index + * + * Lightweight summaries (~5% of the full dataset) embedded as a literal so + * client bundles carry only these small objects — no full evaluation JSON. + * Consumed by lib/client-data.ts. + * + * 196 entities: 68 models, 67 agents, 61 MCP servers. + */ + +import type { EntitySummary } from '@/lib/summary-types'; + +export const ALL_SUMMARIES: EntitySummary[] = [ + { + "id": "claude-fable-5", + "type": "model", + "name": "Claude Fable 5", + "provider": "Anthropic", + "description": "Anthropic's top-tier model above Opus and the most capable widely released Mythos-class model. State-of-the-art on nearly all tested benchmarks at launch, including the highest frontier score on Cognition's FrontierCode. Adaptive thinking only, 1M context, 128K output. Access was suspended globally 2026-06-12 under a US export-control directive after a reported safeguard bypass, and restored 2026-07-01 with a strengthened safety classifier.", + "tags": [ + "coding", + "reasoning", + "enterprise", + "hipaa-eligible", + "safety-focused", + "effort-parameter", + "adaptive-thinking", + "long-context", + "mythos-class", + "flagship" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 92, + "dimensions": { + "performance_reliability": 96, + "security": 90, + "privacy_compliance": 93, + "trust_transparency": 88, + "operational_excellence": 91 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "claude-haiku-4-5", + "type": "model", + "name": "Claude Haiku 4.5", + "provider": "Anthropic", + "description": "Anthropic's fastest model, released October 2025 and still the speed tier of the current lineup (Active as of 2026-07). At launch it beat the since-retired Sonnet 4 on coding (73.3% vs 72.7% SWE-bench) at 1/3 the cost and 2x the speed. First Haiku with extended thinking.", + "tags": [ + "coding", + "fast", + "budget", + "hipaa-eligible", + "privacy", + "extended-thinking", + "computer-use", + "multi-agent", + "value" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 90, + "dimensions": { + "performance_reliability": 93, + "security": 86, + "privacy_compliance": 91, + "trust_transparency": 85, + "operational_excellence": 93 + }, + "strengths_count": 7, + "limitations_count": 5 + }, + { + "id": "claude-opus-4-1", + "type": "model", + "name": "Claude Opus 4.1", + "provider": "Anthropic", + "description": "DEPRECATED: Anthropic deprecated Claude Opus 4.1 (claude-opus-4-1-20250805) on 2026-06-05; it will be retired on 2026-08-05 and requests will then fail. Recommended replacement: Claude Opus 4.8. Historically a flagship model with state-of-the-art reasoning, ASL-3 safety level, and exceptional performance on complex tasks.", + "tags": [ + "deprecated", + "highest-reasoning", + "asl-3-safety", + "hipaa-eligible", + "mission-critical", + "premium" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 91, + "dimensions": { + "performance_reliability": 96, + "security": 92, + "privacy_compliance": 93, + "trust_transparency": 90, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "claude-opus-4-5", + "type": "model", + "name": "Claude Opus 4.5", + "provider": "Anthropic", + "description": "SUPERSEDED: no longer Anthropic's most capable model — succeeded by Opus 4.6, 4.7, 4.8 (2026-05-28) and Claude Fable 5 (2026-06-09, new top tier). At launch it scored 80.9% SWE-bench and was the first model to exceed 80% on SWE-bench Verified, with a unique effort parameter for compute control.", + "tags": [ + "superseded", + "coding", + "reasoning", + "enterprise", + "hipaa-eligible", + "safety-focused", + "effort-parameter", + "computer-use" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 92, + "dimensions": { + "performance_reliability": 96, + "security": 90, + "privacy_compliance": 92, + "trust_transparency": 89, + "operational_excellence": 91 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "claude-opus-4-6", + "type": "model", + "name": "Claude Opus 4.6", + "provider": "Anthropic", + "description": "Anthropic's frontier Opus released February 2026 with 80.8% SWE-bench Verified, breakthrough 68.8% ARC-AGI-2 abstract reasoning, adaptive thinking, and a 1M token context window. Now two generations behind Opus 4.8 but still served.", + "tags": [ + "coding", + "reasoning", + "enterprise", + "hipaa-eligible", + "safety-focused", + "adaptive-thinking", + "effort-parameter", + "computer-use", + "long-context", + "previous-generation" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 92, + "dimensions": { + "performance_reliability": 97, + "security": 91, + "privacy_compliance": 93, + "trust_transparency": 89, + "operational_excellence": 91 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "claude-opus-4-7", + "type": "model", + "name": "Claude Opus 4.7", + "provider": "Anthropic", + "description": "Previous-generation Opus flagship, superseded by Opus 4.8. 64.3% SWE-Bench Pro and 94.2% GPQA Diamond at launch. First Claude with high-resolution vision (2576px long edge, pixel-accurate coordinates), task budgets (beta), and the xhigh effort level.", + "tags": [ + "coding", + "reasoning", + "vision", + "enterprise", + "hipaa-eligible", + "safety-focused", + "effort-parameter", + "adaptive-thinking", + "long-context", + "previous-generation" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 91, + "dimensions": { + "performance_reliability": 95, + "security": 91, + "privacy_compliance": 93, + "trust_transparency": 87, + "operational_excellence": 91 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "claude-opus-4-8", + "type": "model", + "name": "Claude Opus 4.8", + "provider": "Anthropic", + "description": "Anthropic's flagship Opus model with state-of-the-art long-horizon agentic execution, knowledge work, and memory. 84% on Online-Mind2Web, dynamic multi-subagent workflows, ~4x less likely to miss its own code flaws than its predecessor, and 1M context at standard pricing.", + "tags": [ + "coding", + "reasoning", + "agentic", + "enterprise", + "hipaa-eligible", + "safety-focused", + "effort-parameter", + "adaptive-thinking", + "long-context", + "memory", + "flagship" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 92, + "dimensions": { + "performance_reliability": 96, + "security": 92, + "privacy_compliance": 93, + "trust_transparency": 88, + "operational_excellence": 91 + }, + "strengths_count": 7, + "limitations_count": 5 + }, + { + "id": "claude-opus-4", + "type": "model", + "name": "Claude Opus 4", + "provider": "Anthropic", + "description": "RETIRED: Anthropic retired Claude Opus 4 (claude-opus-4-20250514) on 2026-06-15 (deprecated 2026-04-14); API requests now fail. Recommended replacement: Claude Opus 4.8. Historically Anthropic's most powerful model of May 2025, with exceptional reasoning, coding (72.5-79.4% SWE-bench in high-compute), and agentic capabilities.", + "tags": [ + "retired", + "coding", + "reasoning", + "hipaa-eligible", + "privacy", + "enterprise", + "extended-thinking", + "asl-3", + "research" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 95, + "security": 91, + "privacy_compliance": 93, + "trust_transparency": 89, + "operational_excellence": 78 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "claude-sonnet-4-5", + "type": "model", + "name": "Claude Sonnet 4.5", + "provider": "Anthropic", + "description": "Previous-generation Sonnet released September 2025, since superseded by Sonnet 4.6 (2026) and Claude Sonnet 5 (2026-06-30). Still Active on the API with tentative retirement not sooner than 2026-09-29. Historically the top coding model of its era (77.2% SWE-bench Verified at launch) with extended thinking and strong safety features.", + "tags": [ + "coding", + "reasoning", + "enterprise", + "hipaa-eligible", + "safety-focused", + "extended-thinking", + "previous-generation" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 90, + "dimensions": { + "performance_reliability": 94, + "security": 88, + "privacy_compliance": 91, + "trust_transparency": 87, + "operational_excellence": 90 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "claude-sonnet-4-6", + "type": "model", + "name": "Claude Sonnet 4.6", + "provider": "Anthropic", + "description": "Anthropic's previous-generation Sonnet workhorse, superseded by Claude Sonnet 5 (2026-06-30) as the best speed/intelligence balance at the same $3/$15 price. Still fully supported (Active, tentative retirement not sooner than 2027-02-17), with a 1M token context window, 128K max output, adaptive thinking, the effort parameter including 'max', and strong computer-use accuracy.", + "tags": [ + "coding", + "agentic", + "production", + "enterprise", + "hipaa-eligible", + "adaptive-thinking", + "effort-parameter", + "computer-use", + "long-context", + "value-workhorse", + "previous-generation" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 91, + "dimensions": { + "performance_reliability": 93, + "security": 91, + "privacy_compliance": 93, + "trust_transparency": 88, + "operational_excellence": 91 + }, + "strengths_count": 7, + "limitations_count": 5 + }, + { + "id": "claude-sonnet-4", + "type": "model", + "name": "Claude Sonnet 4", + "provider": "Anthropic", + "description": "RETIRED: Anthropic retired Claude Sonnet 4 (claude-sonnet-4-20250514) on 2026-06-15 (deprecated 2026-04-14); API requests now fail. Recommended replacement: Claude Sonnet 4.6. Historically a May 2025 hybrid model with exceptional coding capabilities, advanced reasoning, and extended thinking mode.", + "tags": [ + "retired", + "coding", + "hipaa-eligible", + "privacy", + "enterprise", + "extended-thinking", + "hybrid-reasoning", + "developer-focused" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 88, + "dimensions": { + "performance_reliability": 92, + "security": 90, + "privacy_compliance": 92, + "trust_transparency": 87, + "operational_excellence": 78 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "claude-sonnet-5", + "type": "model", + "name": "Claude Sonnet 5", + "provider": "Anthropic", + "description": "Anthropic's Sonnet-tier flagship (released 2026-06-30), positioned as the cheaper way to run agents: 72.7% SWE-bench Verified (vs Sonnet 4.6's 62.3%), 80.4% Terminal-Bench 2.1, 81.2% OSWorld-Verified — approaching Opus 4.8 at a fraction of the cost. Same $3/$15 standard price as Sonnet 4.6 with intro $2/$10 through 2026-08-31, 1M context, effort parameter incl. xhigh, and deliberately low cyber capability with default-on safeguards.", + "tags": [ + "coding", + "agentic", + "production", + "enterprise", + "hipaa-eligible", + "effort-parameter", + "computer-use", + "long-context", + "value-workhorse", + "new-release" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 91, + "dimensions": { + "performance_reliability": 94, + "security": 91, + "privacy_compliance": 93, + "trust_transparency": 88, + "operational_excellence": 90 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "command-a-plus", + "type": "model", + "name": "Command A+", + "provider": "Cohere", + "description": "Cohere's Apache 2.0 open-weight 218B sparse MoE (25B active) unifying Command A, A Reasoning, A Vision, and A Translate. Runs on 2xH100 or a single B200, supports 48 languages, and ships native citations with grounding spans for verifiable RAG.", + "tags": [ + "enterprise", + "rag", + "citations", + "open-source", + "apache-2-0", + "multilingual", + "multimodal", + "mixture-of-experts", + "vpc-deployment", + "self-hostable" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 87, + "dimensions": { + "performance_reliability": 89, + "security": 87, + "privacy_compliance": 89, + "trust_transparency": 84, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "deepseek-r1", + "type": "model", + "name": "DeepSeek-R1", + "provider": "DeepSeek", + "description": "DeepSeek's standalone reasoning model, now superseded and discontinued. Its reasoning was folded into DeepSeek V3.1's hybrid thinking mode (Aug 2025), then V3.2 (Dec 2025) and V4 (Apr 2026); a successor 'R2' never shipped. The legacy deepseek-reasoner name no longer serves R1 — it routes to V4-Flash's thinking mode and is removed 2026-07-24 (15:59 UTC). R1 remains available only via third-party hosts or self-hosted open weights. Historically 53.6% SWE-bench and 79.8% HumanEval.", + "tags": [ + "superseded", + "coding", + "reasoning", + "open-source", + "cost-effective", + "mathematical", + "chinese-provider", + "value-pricing" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 91, + "security": 83, + "privacy_compliance": 82, + "trust_transparency": 81, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "deepseek-v3-0324", + "type": "model", + "name": "DeepSeek V3 0324", + "provider": "DeepSeek", + "description": "DeepSeek's March 2025 open-weights V3 checkpoint, now superseded by DeepSeek V3.1 (Aug 2025), V3.2 (Dec 2025), and V4 (Apr 2026). The legacy deepseek-chat name no longer serves this checkpoint — it routes to V4-Flash's non-thinking mode and is removed 2026-07-24 (15:59 UTC). V3-0324 remains available only via third-party hosts or self-hosted open weights. Historically offered strong performance at competitive pricing with transparent weights and commercial-friendly licensing.", + "tags": [ + "superseded", + "open-weights", + "cost-effective", + "chinese", + "moe", + "competitive-pricing", + "openai-compatible" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 79, + "security": 76, + "privacy_compliance": 80, + "trust_transparency": 83, + "operational_excellence": 79 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "deepseek-v3-2", + "type": "model", + "name": "DeepSeek-V3.2", + "provider": "DeepSeek", + "description": "DeepSeek's ~685B-parameter MoE flagship with DeepSeek Sparse Attention (DSA) for dramatically cheaper long-context inference. The V3.2-Speciale variant reached IMO 2025 gold-medal level (35/42) and 96.0% AIME. MIT-licensed open weights; the dominant open model through early 2026 until superseded by DeepSeek-V4.", + "tags": [ + "open-source", + "mit-license", + "reasoning", + "sparse-attention", + "long-context", + "cost-effective", + "mathematical", + "chinese-provider", + "superseded-by-v4" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 93, + "security": 83, + "privacy_compliance": 78, + "trust_transparency": 82, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "deepseek-v4", + "type": "model", + "name": "DeepSeek-V4", + "provider": "DeepSeek", + "description": "DeepSeek's preview flagship family: V4-Pro (1.6T total / 49B active MoE, largest open-weight release ever) and V4-Flash (284B/13B). 1M context, up to 384K output, via manifold-constrained Hyper Connections and Constrained Sparse Attention. MIT license. Vendor benchmarks await broad independent verification. Per DeepSeek (2026-06-30), V4 graduates to official release mid-July 2026 — same model names, with peak-hour API pricing (2x baseline, Beijing 9:00-12:00 and 14:00-18:00).", + "tags": [ + "open-source", + "mit-license", + "preview", + "long-context", + "reasoning", + "sparse-attention", + "cost-effective", + "chinese-provider", + "flagship" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 92, + "security": 83, + "privacy_compliance": 78, + "trust_transparency": 80, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "gemini-2-0-flash", + "type": "model", + "name": "Gemini 2.0 Flash", + "provider": "Google", + "description": "SHUT DOWN: Google shut down the Gemini 2.0 Flash family on 2026-06-01; the model is no longer served via the Gemini API. Historically a fast, efficient multimodal model (53.6% SWE-bench, 62.1% MMLU) optimized for speed and vision. Migrate to Gemini 3.5 Flash for equivalent fast multimodal workloads.", + "tags": [ + "retired", + "fast", + "multimodal", + "vision", + "large-context", + "cost-effective", + "hipaa-eligible", + "real-time", + "google-cloud" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 89, + "security": 87, + "privacy_compliance": 88, + "trust_transparency": 84, + "operational_excellence": 89 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "gemini-2-5-pro", + "type": "model", + "name": "Gemini 2.5 Pro", + "provider": "Google", + "description": "DEPRECATED: Google has scheduled Gemini 2.5 Pro for shutdown on 2026-10-16; the designated replacement is Gemini 3.1 Pro. Formerly Google's flagship (superseded by the Gemini 3.x line since late 2025), with 1M token context window (2M on select Vertex AI enterprise tiers), Deep Think mode for complex reasoning, and native multimodal capabilities. Do not start new projects on this model.", + "tags": [ + "deprecated", + "long-context", + "1m-tokens", + "deep-think", + "multimodal", + "cost-effective", + "google-cloud" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 93, + "security": 86, + "privacy_compliance": 85, + "trust_transparency": 88, + "operational_excellence": 92 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "gemini-3-1-pro", + "type": "model", + "name": "Gemini 3.1 Pro", + "provider": "Google", + "description": "Google's current flagship reasoning model with 77.1% ARC-AGI-2 (2.5x Gemini 3 Pro), 94.3% GPQA Diamond, 2887 Elo on LiveCodeBench Pro, and 1M token context. Supersedes the retired Gemini 3 Pro Preview (shut down 2026-03-09; the gemini-pro-latest alias now points here). Note: still served under the preview model ID gemini-3.1-pro-preview — official docs do not list it as GA, contrary to earlier reports. Gemini 3.5 Pro (announced I/O May 2026) has not shipped as of 2026-07-09.", + "tags": [ + "flagship", + "preview", + "reasoning", + "long-context", + "1m-tokens", + "multimodal", + "google-cloud", + "enterprise" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 91, + "dimensions": { + "performance_reliability": 96, + "security": 88, + "privacy_compliance": 88, + "trust_transparency": 88, + "operational_excellence": 93 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "gemini-3-5-flash", + "type": "model", + "name": "Gemini 3.5 Flash", + "provider": "Google", + "description": "Google's GA 'frontier workhorse' launched at I/O 2026. Beats Gemini 3.1 Pro on agentic and coding suites (76.2% Terminal-Bench 2.1, 83.6% MCP Atlas) at roughly 4x the speed, with 1M token context. Pricier than past Flash tiers at $1.50/$9.00 per 1M.", + "tags": [ + "workhorse", + "ga", + "agentic", + "fast", + "long-context", + "1m-tokens", + "multimodal", + "google-cloud" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 89, + "dimensions": { + "performance_reliability": 93, + "security": 87, + "privacy_compliance": 88, + "trust_transparency": 86, + "operational_excellence": 93 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "gemini-3-flash", + "type": "model", + "name": "Gemini 3 Flash", + "provider": "Google", + "description": "Google's efficiency model with Pro-level performance at low cost. 78% SWE-bench (beat Gemini 3 Pro), 1M context, 3x faster than 2.5 Pro. Thinking level parameter for compute control. Still served as gemini-3-flash-preview (no shutdown date announced), but superseded as Google's lead Flash tier by Gemini 3.5 Flash (2026-05-19), which Google lists as its designated replacement.", + "tags": [ + "superseded", + "cost-effective", + "fast", + "1m-tokens", + "multimodal", + "thinking-level", + "free-tier", + "high-volume", + "value-leader" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 93, + "security": 86, + "privacy_compliance": 85, + "trust_transparency": 88, + "operational_excellence": 92 + }, + "strengths_count": 7, + "limitations_count": 5 + }, + { + "id": "gemini-3-pro", + "type": "model", + "name": "Gemini 3 Pro", + "provider": "Google", + "description": "SHUT DOWN: Google retired Gemini 3 Pro Preview on 2026-03-09 (the gemini-pro-latest alias moved to 3.1 Pro on 2026-03-06); it is no longer served via the Gemini API or AI Studio. Historically a former Google flagship with 1M token context, 1501 LMArena Elo (first model >1500), Deep Think mode, and native multimodal. Migrate to Gemini 3.1 Pro (same pricing, ARC-AGI-2 77.1% vs 31.1%).", + "tags": [ + "retired", + "long-context", + "1m-tokens", + "deep-think", + "multimodal", + "google-cloud" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 90, + "dimensions": { + "performance_reliability": 95, + "security": 87, + "privacy_compliance": 86, + "trust_transparency": 90, + "operational_excellence": 91 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "gemma-3-27b", + "type": "model", + "name": "Gemma 3 27B", + "provider": "Google", + "description": "Google's open-source Gemma 3 model with 27 billion parameters, now superseded by Gemma 4 (released 2026-04-02 under Apache 2.0, a license improvement over the custom Gemma license). Was designed for developers seeking Google's research quality with open-source flexibility; new deployments should evaluate Gemma 4 instead.", + "tags": [ + "superseded", + "open-source", + "google", + "privacy", + "basic", + "commercial-friendly", + "lightweight" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 70, + "security": 78, + "privacy_compliance": 94, + "trust_transparency": 85, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "gemma-4", + "type": "model", + "name": "Gemma 4", + "provider": "Google", + "description": "Google's open-weight family released April 2026 under Apache 2.0 (a shift from the custom Gemma license). Spans E2B/E4B edge models with 128K context and native audio up to a 31B dense model with 256K context. The 31B scores ~1452 on LMArena, No. 3 among open models.", + "tags": [ + "open-source", + "apache-2-0", + "open-weights", + "edge", + "on-device", + "moe", + "multimodal", + "self-hosted", + "data-sovereignty" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 81, + "dimensions": { + "performance_reliability": 82, + "security": 78, + "privacy_compliance": 84, + "trust_transparency": 80, + "operational_excellence": 80 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "glm-5-2", + "type": "model", + "name": "GLM-5.2", + "provider": "Z.ai (Zhipu AI)", + "description": "Z.ai's MIT-licensed 744B-parameter MoE (40B active) launched June 2026 with a 1M-token context via IndexShare sparse attention. Leading open-weight model on Artificial Analysis Intelligence Index v4.1 (51), with 62.1 SWE-bench Pro, 81.0 Terminal-Bench 2.1, and 99.2% AIME 2026 at $1.40/$4.40 per 1M tokens. Weights published 2026-06-16.", + "tags": [ + "coding", + "reasoning", + "open-source", + "mit-license", + "mixture-of-experts", + "agentic", + "long-context", + "cost-effective", + "chinese-provider", + "self-hostable" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 82, + "dimensions": { + "performance_reliability": 93, + "security": 79, + "privacy_compliance": 75, + "trust_transparency": 80, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "glm-5", + "type": "model", + "name": "GLM-5", + "provider": "Z.ai (Zhipu AI)", + "description": "Z.ai's MIT-licensed 744B-parameter MoE (40B active) with 77.8% SWE-bench Verified, 92.7% AIME 2026, and open-source leadership on BrowseComp and agentic benchmarks. Trained on 28.5T tokens with DeepSeek Sparse Attention.", + "tags": [ + "coding", + "reasoning", + "open-source", + "mit-license", + "mixture-of-experts", + "agentic", + "cost-effective", + "chinese-provider", + "self-hostable" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 82, + "dimensions": { + "performance_reliability": 92, + "security": 80, + "privacy_compliance": 75, + "trust_transparency": 81, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "gpt-4-1-mini", + "type": "model", + "name": "GPT-4.1 mini", + "provider": "OpenAI", + "description": "LEGACY: retired from ChatGPT 2026-02-13 but still available in the API with no announced shutdown (as of 2026-07-09). Balanced GPT-4.1 variant with a 1,047,576-token context window, offering good performance at reasonable cost. OpenAI recommends GPT-5.x mini tiers for new work.", + "tags": [ + "balanced", + "production-ready", + "cost-effective", + "general-purpose", + "fast", + "mid-tier" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 78, + "security": 84, + "privacy_compliance": 84, + "trust_transparency": 80, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "gpt-4-1-nano", + "type": "model", + "name": "GPT-4.1 nano", + "provider": "OpenAI", + "description": "DEPRECATED: OpenAI announced 2026-04-22 that gpt-4.1-nano's API shuts down 2026-10-23; recommended replacement is gpt-5.4-nano. Historically OpenAI's smallest and most efficient GPT-4.1 variant for high-volume, cost-sensitive applications, with a 1,047,576-token context window.", + "tags": [ + "deprecated", + "efficient", + "low-latency", + "cost-effective", + "basic", + "high-volume", + "real-time" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 68, + "security": 82, + "privacy_compliance": 84, + "trust_transparency": 76, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "gpt-4-1", + "type": "model", + "name": "GPT-4.1", + "provider": "OpenAI", + "description": "LEGACY: retired from ChatGPT 2026-02-13 but still available in the API with no announced shutdown (as of 2026-07-09). Previous-generation general-purpose GPT-4.1 model with a 1,047,576-token context window. OpenAI recommends GPT-5.x models for new work.", + "tags": [ + "general-purpose", + "flagship", + "production-ready", + "multimodal", + "enterprise", + "balanced" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 85, + "security": 86, + "privacy_compliance": 84, + "trust_transparency": 82, + "operational_excellence": 90 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "gpt-4o-mini", + "type": "model", + "name": "GPT-4o mini", + "provider": "OpenAI", + "description": "OpenAI's efficient multimodal model combining text and vision capabilities at competitive pricing. Designed for cost-sensitive applications requiring basic multimodal understanding. Legacy model: remains available in the API with no announced shutdown as of 2026-07-09 (unaffected by the GPT-4o retirement), but OpenAI recommends newer GPT-5.x mini/nano tiers for new work.", + "tags": [ + "multimodal", + "vision", + "cost-effective", + "fast", + "image-understanding", + "ocr" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 81, + "dimensions": { + "performance_reliability": 71, + "security": 83, + "privacy_compliance": 84, + "trust_transparency": 81, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "gpt-4o", + "type": "model", + "name": "GPT-4o", + "provider": "OpenAI", + "description": "DEPRECATED: removed from ChatGPT 2026-02-13 and fully retired from ChatGPT (including Custom GPTs) 2026-04-03; chatgpt-4o-latest API access ended 2026-02-16; the gpt-4o-2024-05-13 snapshot's API shuts down 2026-10-23 (gpt-4o-2024-11-20 remains served via API). Historically OpenAI's flagship multimodal model with strong text and vision capabilities for high-quality multimodal understanding and generation. Migrate to newer GPT-5.x models.", + "tags": [ + "deprecated", + "multimodal", + "vision", + "image-understanding", + "ocr", + "education" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 81, + "security": 85, + "privacy_compliance": 84, + "trust_transparency": 82, + "operational_excellence": 87 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "gpt-5-1", + "type": "model", + "name": "GPT-5.1", + "provider": "OpenAI", + "description": "SUPERSEDED by GPT-5.2/5.4/5.5 and the GPT-5.6 family (released 2026-07-09); retired from ChatGPT 2026-03-11. gpt-5.1-chat-latest, gpt-5.1-codex, gpt-5.1-codex-max, and gpt-5.1-codex-mini shut down in the API 2026-07-23 (migrate to gpt-5.5 / gpt-5.4-mini); the base gpt-5.1 snapshot remains served with no announced shutdown. Released Nov 2025 with adaptive reasoning (2-3x faster on simple tasks), 76.3% SWE-bench, developer tools (apply_patch, shell), warmer tone.", + "tags": [ + "superseded", + "general-purpose", + "multimodal", + "low-latency", + "ecosystem-leader", + "unified-thinking", + "audio-capable" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 95, + "security": 85, + "privacy_compliance": 84, + "trust_transparency": 89, + "operational_excellence": 93 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "gpt-5-2-codex", + "type": "model", + "name": "GPT-5.2 Codex", + "provider": "OpenAI", + "description": "DEPRECATED: superseded by GPT-5.3-Codex (2026-02-05); gpt-5.2-codex API shuts down 2026-07-23 (two weeks from 2026-07-09 — migrate now; OpenAI's listed replacement is gpt-5.5). Historically OpenAI's specialized coding model built on GPT-5.2 with 56.4% SWE-bench Pro, 64% Terminal-bench 2.0, native code compaction, and enhanced cybersecurity capabilities.", + "tags": [ + "deprecated", + "coding", + "specialized", + "swe-bench-leader", + "terminal", + "cybersecurity", + "code-compaction", + "developer-tools" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 90, + "dimensions": { + "performance_reliability": 96, + "security": 89, + "privacy_compliance": 85, + "trust_transparency": 88, + "operational_excellence": 91 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "gpt-5-2", + "type": "model", + "name": "GPT-5.2", + "provider": "OpenAI", + "description": "SUPERSEDED by GPT-5.4 (2026-03-05), GPT-5.5 (2026-04-23), and the GPT-5.6 family (2026-07-09). API snapshots (gpt-5.2, gpt-5.2-2025-12-11) still served with no announced shutdown, but gpt-5.2-chat-latest shuts down 2026-08-10 (announced 2026-05-08; migrate to gpt-5.5). 400K context window, 100% AIME 2025 score, 52.9% ARC-AGI-2. Three variants: Instant (speed), Thinking (reasoning), Pro (accuracy). New projects should prefer GPT-5.5.", + "tags": [ + "superseded", + "reasoning", + "multimodal", + "400k-context", + "ecosystem-leader", + "math-expert", + "three-variants", + "low-latency" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 91, + "dimensions": { + "performance_reliability": 97, + "security": 87, + "privacy_compliance": 85, + "trust_transparency": 91, + "operational_excellence": 95 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "gpt-5-3-codex", + "type": "model", + "name": "GPT-5.3-Codex", + "provider": "OpenAI", + "description": "OpenAI's agentic coding specialist, still active with no announced shutdown (as of 2026-07-09): ~80% SWE-bench Verified, 77.3% Terminal-Bench, SOTA on SWE-Bench Pro at release, ~25% faster than GPT-5.2-Codex. The 5.3 generation shipped no general-purpose API flagship — only this Codex model plus a gpt-5.3-chat-latest ChatGPT alias that shuts down 2026-08-10.", + "tags": [ + "coding", + "agentic", + "codex", + "specialist", + "terminal", + "swe-bench-leader", + "cost-efficient" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 92, + "security": 87, + "privacy_compliance": 87, + "trust_transparency": 86, + "operational_excellence": 93 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "gpt-5-4", + "type": "model", + "name": "GPT-5.4", + "provider": "OpenAI", + "description": "OpenAI's previous-generation flagship (superseded by GPT-5.5 in April 2026 and the GPT-5.6 family in July 2026). Fully supported with no announced shutdown; gpt-5.4-mini and gpt-5.4-nano are OpenAI's designated replacements for retiring GPT-5 mini/nano and gpt-4.1-nano. Headline native computer use with 75% OSWorld-Verified, ~33% fewer factual errors than GPT-5.2, ~1.05M context. Thinking, Pro, mini, and nano variants.", + "tags": [ + "computer-use", + "previous-flagship", + "million-token-context", + "factuality", + "variant-family", + "agentic", + "multimodal" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 91, + "dimensions": { + "performance_reliability": 95, + "security": 88, + "privacy_compliance": 87, + "trust_transparency": 89, + "operational_excellence": 94 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "gpt-5-5", + "type": "model", + "name": "GPT-5.5", + "provider": "OpenAI", + "description": "OpenAI's flagship from April 2026 (codename 'Spud'), first fully retrained base model since GPT-4.5. Succeeded at the top of the lineup by the GPT-5.6 family (Sol/Terra/Luna, publicly released 2026-07-09) but fully supported with no announced shutdown, and still the designated migration target for most of the GPT-5.x line. ~1.05M context, 85.0% ARC-AGI-2, 93.6% GPQA Diamond, 58.6% SWE-Bench Pro.", + "tags": [ + "reasoning", + "flagship", + "million-token-context", + "agentic", + "computer-use", + "retrained-base", + "migration-target", + "token-efficient" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 91, + "dimensions": { + "performance_reliability": 97, + "security": 89, + "privacy_compliance": 87, + "trust_transparency": 90, + "operational_excellence": 94 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "gpt-5-6", + "type": "model", + "name": "GPT-5.6", + "provider": "OpenAI", + "description": "OpenAI's GPT-5.6 family — Sol (flagship, $5/$30), Terra (balanced, $2.50/$15), Luna (fast, $1/$6 per 1M) — previewed 2026-06-26 under US-government-requested partner-only restrictions and publicly released 2026-07-09. Sol posts 88.8% Terminal-Bench 2.1 (91.9% in Ultra mode); Terra is reported GPT-5.5-class at half the price. ~1.5M context reported but unconfirmed. Launch-day evaluation: independent verification is still very limited.", + "tags": [ + "flagship", + "variant-family", + "agentic", + "reasoning", + "launch-day", + "government-preview", + "tiered-pricing", + "token-efficient" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 89, + "dimensions": { + "performance_reliability": 94, + "security": 88, + "privacy_compliance": 87, + "trust_transparency": 87, + "operational_excellence": 90 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "gpt-5", + "type": "model", + "name": "GPT-5", + "provider": "OpenAI", + "description": "DEPRECATED: retired from ChatGPT (GPT-5 Instant/Thinking removed by 2026-02-13); gpt-5-chat-latest and gpt-5-codex API shut down 2026-07-23; the gpt-5-2025-08-07 snapshot's API shuts down 2026-12-11 (announced 2026-06-11) — migrate to GPT-5.5. Historically OpenAI's flagship model with unified thinking capabilities, multimodal understanding, and enhanced reasoning; successor to the GPT-4o series.", + "tags": [ + "deprecated", + "general-purpose", + "multimodal", + "low-latency", + "ecosystem-leader", + "unified-thinking", + "audio-capable" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 95, + "security": 85, + "privacy_compliance": 84, + "trust_transparency": 89, + "operational_excellence": 93 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "gpt-oss-120b", + "type": "model", + "name": "GPT-OSS-120B", + "provider": "OpenAI", + "description": "OpenAI's first open-weight model released August 2025. 117B total params (5.1B active), Apache 2.0 license. Matches o4-mini on many benchmarks. Runs in 80GB memory.", + "tags": [ + "open-source", + "apache-2.0", + "self-hosted", + "privacy", + "moe", + "reasoning", + "coding", + "on-premises", + "customizable", + "transparent" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 92, + "dimensions": { + "performance_reliability": 91, + "security": 85, + "privacy_compliance": 97, + "trust_transparency": 94, + "operational_excellence": 95 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "gpt-oss-20b", + "type": "model", + "name": "GPT-OSS-20B", + "provider": "OpenAI", + "description": "OpenAI's edge-optimized open-weight model released August 2025. 21B total params (3.6B active), Apache 2.0 license. Matches o3-mini despite small size. Runs in 16GB memory (edge devices).", + "tags": [ + "open-source", + "apache-2.0", + "self-hosted", + "privacy", + "moe", + "reasoning", + "coding", + "on-premises", + "customizable", + "transparent" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 92, + "dimensions": { + "performance_reliability": 91, + "security": 85, + "privacy_compliance": 97, + "trust_transparency": 94, + "operational_excellence": 95 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "grok-3-beta", + "type": "model", + "name": "Grok 3 [Beta]", + "provider": "xAI", + "description": "RETIRED: xAI retired Grok 3 on 2026-05-15; retired API slugs now silently redirect to Grok 4.3 at Grok 4.3 pricing. Historically xAI's flagship beta model with exceptional coding performance and real-time knowledge via X platform. Migrate to Grok 4.3 or the new flagship Grok 4.5 (released 2026-07-08). Note: xAI merged into SpaceX and rebranded as SpaceXAI in mid-2026.", + "tags": [ + "retired", + "beta", + "coding", + "real-time", + "x-integration", + "cutting-edge", + "high-performance" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 94, + "security": 84, + "privacy_compliance": 78, + "trust_transparency": 83, + "operational_excellence": 78 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "grok-4-1", + "type": "model", + "name": "Grok 4.1", + "provider": "xAI", + "description": "xAI's late-2025 flagship that debuted #1 on LMArena Text (1483 Elo) and led EQ-Bench3 for emotional intelligence, with a 2M token context window. Now two generations behind: superseded by Grok 4.3 (2026-04-30) and the new flagship Grok 4.5 (2026-07-08). The grok-4-1-fast variants were retired on 2026-05-15; xAI itself merged into SpaceX and rebranded as SpaceXAI in mid-2026.", + "tags": [ + "superseded", + "long-context", + "emotional-intelligence", + "lmarena-leader", + "reasoning", + "legacy" + ], + "last_evaluated": "2026-07-09", + "release_year": 2025, + "overall_score": 83, + "dimensions": { + "performance_reliability": 92, + "security": 83, + "privacy_compliance": 76, + "trust_transparency": 82, + "operational_excellence": 81 + }, + "strengths_count": 5, + "limitations_count": 5 + }, + { + "id": "grok-4-3", + "type": "model", + "name": "Grok 4.3", + "provider": "xAI", + "description": "xAI's workhorse model (released 2026-04-30): 1M context, reasoning, function calling, and structured outputs at $1.25/$2.50 per 1M tokens. Superseded as flagship by Grok 4.5 (2026-07-08, $2/$6, 500K context) but remains served and is the redirect target for retired Grok slugs. Strong frontier performance, but thinner enterprise compliance than Anthropic/OpenAI/Google, and the provider (now SpaceXAI post-SpaceX merger) faces active regulatory investigations over Grok content safety.", + "tags": [ + "reasoning", + "long-context", + "function-calling", + "structured-outputs", + "cost-effective", + "real-time" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 83, + "dimensions": { + "performance_reliability": 94, + "security": 82, + "privacy_compliance": 76, + "trust_transparency": 81, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "grok-4-5", + "type": "model", + "name": "Grok 4.5", + "provider": "SpaceXAI (formerly xAI)", + "description": "SpaceXAI's flagship (released 2026-07-08, days after the xAI-to-SpaceXAI rebrand): an 'Opus-class' model tuned for token efficiency at $2/$6 per 1M ($0.50 cached), 500K context, reasoning effort low/medium/high. Strong day-one results (83.3% Terminal-Bench 2.1, #4 on Artificial Analysis Intelligence Index) but no EU availability at launch, thin enterprise compliance, and a provider under active regulatory investigation over Grok content-safety failures.", + "tags": [ + "reasoning", + "token-efficient", + "cost-effective", + "function-calling", + "structured-outputs", + "real-time", + "new-release", + "regulatory-scrutiny", + "no-eu-availability" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 82, + "dimensions": { + "performance_reliability": 93, + "security": 82, + "privacy_compliance": 75, + "trust_transparency": 80, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "kimi-k2-6", + "type": "model", + "name": "Kimi K2.6", + "provider": "Moonshot AI", + "description": "Moonshot AI's open-weight 1T-parameter MoE (32B active) with vendor-reported 80.2% SWE-Bench Verified and 58.6 SWE-Bench Pro. Agent Swarm orchestration scales to 300 sub-agents and 4,000 coordinated steps for long-horizon coding. Remains Moonshot's general-purpose flagship as of July 2026; a coding-specialized sibling, Kimi K2.7-Code (built on K2.6, also open-weight Modified MIT), shipped 2026-06-12.", + "tags": [ + "coding", + "agentic", + "open-source", + "mixture-of-experts", + "long-context", + "agent-swarm", + "chinese-provider", + "self-hostable" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 81, + "dimensions": { + "performance_reliability": 91, + "security": 79, + "privacy_compliance": 75, + "trust_transparency": 80, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "kimi-k2-7-code", + "type": "model", + "name": "Kimi K2.7-Code", + "provider": "Moonshot AI", + "description": "Moonshot AI's coding-specialized open-weight MoE (1T total / 32B active, Modified MIT) built on Kimi K2.6, released 2026-06-12. Vendor reports 62.0 on Kimi Code Bench v2 (+21.8% over K2.6) and ~30% lower reasoning-token usage, but all published benchmarks are Moonshot-run with no independent public-suite results yet. 256K context, thinking mode always on, $0.95/$4.00 per 1M tokens.", + "tags": [ + "coding", + "agentic", + "open-source", + "mixture-of-experts", + "long-context", + "token-efficient", + "chinese-provider", + "self-hostable" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 80, + "dimensions": { + "performance_reliability": 90, + "security": 78, + "privacy_compliance": 74, + "trust_transparency": 77, + "operational_excellence": 81 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "llama-3-1-405b", + "type": "model", + "name": "Llama 3.1 405B", + "provider": "Meta", + "description": "Meta's largest open-source model with 405 billion parameters, offering complete transparency, self-hosting capabilities, and competitive performance with proprietary models. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models with Muse Spark (April 2026). Weights remain broadly available on Hugging Face and via many API hosts as of July 2026.", + "tags": [ + "meta", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 86, + "security": 78, + "privacy_compliance": 96, + "trust_transparency": 95, + "operational_excellence": 82 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "llama-3-3-70b", + "type": "model", + "name": "Llama 3.3 70B", + "provider": "Meta", + "description": "Meta's powerful 70B parameter Llama 3.3 model offering strong performance with open-source flexibility and an excellent balance of capability and resource efficiency for self-hosted deployments. Remains one of Meta's legacy open models: Meta has shipped no new open weights since Llama 4 Scout/Maverick (April 2025) and pivoted to closed models with Muse Spark (April 2026). Weights remain widely available and hosted as of July 2026.", + "tags": [ + "open-source", + "mathematics", + "self-hosted", + "privacy", + "balanced", + "70b-parameters" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 78, + "security": 80, + "privacy_compliance": 95, + "trust_transparency": 87, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "llama-4-behemoth", + "type": "model", + "name": "Llama 4 Behemoth", + "provider": "Meta", + "description": "Meta's announced 2T-total/288B-active parameter Llama 4 teacher model that was NEVER RELEASED. It remains 'announced, not released' as of July 2026 — effectively shelved (never formally cancelled): Meta gave no update when asked in January 2026 and has effectively exited open-weight frontier releases, shipping the proprietary closed-weight 'Muse Spark' (April 8, 2026) instead. Scores reflect unverifiable preview-era claims; the model is not available for any deployment.", + "tags": [ + "unreleased", + "open-source", + "self-hosted", + "mathematics", + "enterprise", + "privacy", + "reasoning", + "large-scale" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 84, + "security": 82, + "privacy_compliance": 95, + "trust_transparency": 88, + "operational_excellence": 77 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "llama-4-maverick", + "type": "model", + "name": "Llama 4 Maverick", + "provider": "Meta", + "description": "Meta's flagship open-weight model (released April 5, 2025): a mixture-of-experts with 400B total / 17B active parameters (128 experts), natively multimodal (text + image), with a 1M-token context window. Meta's last open-weight release: the company has since pivoted to closed models with Muse Spark (April 2026), and newer open models from DeepSeek, Qwen, and Moonshot have surpassed it on most benchmarks.", + "tags": [ + "open-source", + "self-hosted", + "data-sovereignty", + "on-premises", + "customizable", + "fine-tunable", + "cost-effective-at-scale" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 91, + "security": 80, + "privacy_compliance": 95, + "trust_transparency": 94, + "operational_excellence": 85 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "llama-4-scout", + "type": "model", + "name": "Llama 4 Scout", + "provider": "Meta", + "description": "Meta's efficient Llama 4 model (released April 5, 2025): a natively multimodal mixture-of-experts with 109B total / 17B active parameters (16 experts) and an industry-leading 10M-token context window, deployable on a single H100-class GPU. Optimized for speed and cost-sensitive applications requiring open-weight flexibility. Now a legacy line: Meta has shipped no new open weights since Scout/Maverick and pivoted to closed models with Muse Spark (April 2026).", + "tags": [ + "open-source", + "efficient", + "edge-deployment", + "low-latency", + "privacy", + "cost-effective" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 76, + "security": 80, + "privacy_compliance": 95, + "trust_transparency": 86, + "operational_excellence": 86 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "minimax-m2", + "type": "model", + "name": "MiniMax-M2", + "provider": "MiniMax", + "description": "MiniMax's MIT-licensed 230B MoE with only 10B active parameters, optimized for agentic tool calling and coding. Topped open-model agentic rankings at launch and undercut Claude pricing by roughly 92% while remaining fast due to its small active footprint.", + "tags": [ + "agentic", + "tool-calling", + "open-source", + "mit-license", + "mixture-of-experts", + "cost-effective", + "fast-inference", + "chinese-provider", + "self-hostable" + ], + "last_evaluated": "2026-07-09", + "release_year": 2025, + "overall_score": 80, + "dimensions": { + "performance_reliability": 87, + "security": 77, + "privacy_compliance": 74, + "trust_transparency": 78, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "minimax-m3", + "type": "model", + "name": "MiniMax-M3", + "provider": "MiniMax", + "description": "MiniMax's natively multimodal 428B MoE (23B active) with a 1M-token context via MiniMax Sparse Attention, launched June 2026 with weights on Hugging Face by 2026-06-07. Vendor reports 80.5% SWE-bench Verified; Artificial Analysis scores it 44 on Intelligence Index v4.1. Unlike MIT-licensed M2, M3 ships under a MiniMax Community License, and training code and some inference operators are withheld.", + "tags": [ + "agentic", + "multimodal", + "tool-calling", + "open-weight", + "mixture-of-experts", + "long-context", + "cost-effective", + "fast-inference", + "chinese-provider", + "self-hostable" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 78, + "dimensions": { + "performance_reliability": 88, + "security": 76, + "privacy_compliance": 73, + "trust_transparency": 75, + "operational_excellence": 77 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mistral-large-3", + "type": "model", + "name": "Mistral Large 3", + "provider": "Mistral AI", + "description": "Mistral AI's open-weight flagship released December 2025 under Apache 2.0: a sparse MoE (675B total / 41B active) multimodal model with ~256K context and 40+ languages. Debuted #2 among open-source non-reasoning models on LMArena, with a strong EU data-sovereignty story.", + "tags": [ + "open-source", + "apache-2.0", + "mixture-of-experts", + "multilingual", + "eu-sovereignty", + "gdpr", + "self-hostable", + "multimodal" + ], + "last_evaluated": "2026-07-09", + "release_year": 2025, + "overall_score": 85, + "dimensions": { + "performance_reliability": 88, + "security": 83, + "privacy_compliance": 87, + "trust_transparency": 80, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "nemotron-3-ultra", + "type": "model", + "name": "Nemotron 3 Ultra", + "provider": "NVIDIA", + "description": "NVIDIA's open frontier reasoning model, released 2026-06-04 to complete the Nemotron 3 rollout (Nano Dec 2025, Super Mar 2026): a 550B total / 55B active LatentMoE hybrid Mamba-Transformer under OpenMDW-1.1 with open weights, training data, and recipes. 1M-token context, 71.9% SWE-bench Verified (vendor), Artificial Analysis Index 48 — the top-scoring US open-weight model — with ~140 tok/s decode and the best non-hallucination score in its comparison set (78.7 AA-Omniscience).", + "tags": [ + "open-source", + "mixture-of-experts", + "hybrid-mamba-transformer", + "long-context", + "reasoning", + "agentic", + "open-training-data", + "nvidia-ecosystem", + "flagship" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 91, + "security": 82, + "privacy_compliance": 83, + "trust_transparency": 85, + "operational_excellence": 85 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "nemotron-ultra-253b", + "type": "model", + "name": "Nemotron Ultra 253B", + "provider": "NVIDIA", + "description": "253B parameter model from NVIDIA's Llama-3.1-based Nemotron line, now superseded: NVIDIA discontinued this line in favor of the native Nemotron 3 family, whose rollout completed in June 2026 (Nano 2025-12, Super 2026-03, and Nemotron 3 Ultra — a 550B total / 55B active MoE hybrid Mamba-Transformer — on 2026-06-04). Historically 57.1% SWE-bench and 80.08% HumanEval, optimized for HPC and complex coding with GPU acceleration. New deployments should evaluate Nemotron 3 Ultra instead.", + "tags": [ + "superseded", + "coding", + "gpu-accelerated", + "enterprise", + "soc-2-certified", + "large-model", + "nvidia-ecosystem", + "high-performance" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 92, + "security": 85, + "privacy_compliance": 87, + "trust_transparency": 83, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "nova-2-lite", + "type": "model", + "name": "Amazon Nova 2 Lite", + "provider": "Amazon (AWS)", + "description": "Amazon's cost-efficient Nova 2 workhorse model, GA on Amazon Bedrock since re:Invent 2025. Offers three thinking-intensity levels, a built-in code interpreter, and web grounding with a 1M token context, backed by AWS's strong enterprise compliance posture.", + "tags": [ + "aws", + "bedrock", + "enterprise", + "hipaa-eligible", + "cost-effective", + "code-interpreter", + "web-grounding", + "thinking-levels" + ], + "last_evaluated": "2026-07-09", + "release_year": 2025, + "overall_score": 87, + "dimensions": { + "performance_reliability": 86, + "security": 88, + "privacy_compliance": 90, + "trust_transparency": 81, + "operational_excellence": 89 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "nova-pro", + "type": "model", + "name": "Nova Pro", + "provider": "Amazon", + "description": "SUPERSEDED by the Nova 2 family announced at re:Invent 2025-12-02 (Nova 2 Lite GA; Nova 2 Pro/Omni still in preview as of 2026-07-09), though the original Nova Pro is still served. Amazon model integrated with AWS services for enterprise customers requiring seamless AWS integration. New projects should evaluate Nova 2.", + "tags": [ + "superseded", + "aws", + "enterprise", + "hipaa-eligible", + "fedramp", + "compliance", + "integrated" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 86, + "dimensions": { + "performance_reliability": 73, + "security": 89, + "privacy_compliance": 93, + "trust_transparency": 81, + "operational_excellence": 94 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "openai-o1-mini", + "type": "model", + "name": "OpenAI o1-mini", + "provider": "OpenAI", + "description": "RETIRED: o1-mini was shut down in the OpenAI API on 2025-10-27 (o1-preview on 2025-07-28) and removed from ChatGPT; it is no longer available anywhere. OpenAI's stated replacement was o4-mini, itself now scheduled for shutdown 2026-10-23 — migrate new work to the GPT-5.x family (gpt-5.5 / gpt-5.4-mini). Historically an efficient reasoning model with chain-of-thought capabilities at lower cost than o1.", + "tags": [ + "retired", + "reasoning", + "chain-of-thought", + "coding", + "education", + "balanced", + "cost-effective" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 82, + "security": 85, + "privacy_compliance": 84, + "trust_transparency": 84, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "openai-o1", + "type": "model", + "name": "OpenAI o1", + "provider": "OpenAI", + "description": "DEPRECATED: o1 variants and o1-pro shut down in the API on 2026-10-23; already removed from ChatGPT (o1-preview/o1-mini removed in 2025). Migration target is GPT-5.5. Historically an advanced reasoning model (57.1% SWE-bench, 79.2% HumanEval) with extended chain-of-thought reasoning.", + "tags": [ + "deprecated", + "reasoning", + "chain-of-thought", + "coding", + "mathematics", + "research", + "explainable", + "soc-2-certified", + "high-latency" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 93, + "security": 88, + "privacy_compliance": 86, + "trust_transparency": 90, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "openai-o3-mini", + "type": "model", + "name": "OpenAI o3-mini", + "provider": "OpenAI", + "description": "DEPRECATED: o3-mini's API shuts down 2026-10-23; migration target is GPT-5.5. Historically an efficient reasoning model from OpenAI achieving 50% on SWE-bench and 87.3% on HumanEval, optimized for fast reasoning at competitive pricing with strong coding capabilities.", + "tags": [ + "deprecated", + "reasoning", + "code-generation", + "mini-model", + "budget-friendly", + "chain-of-thought", + "efficient", + "soc-2-certified" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 89, + "security": 86, + "privacy_compliance": 84, + "trust_transparency": 87, + "operational_excellence": 87 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "openai-o3", + "type": "model", + "name": "OpenAI o3", + "provider": "OpenAI", + "description": "DEPRECATED: the entire o3 family now has shutdown dates — o3 (o3-2025-04-16) and o3-pro shut down in the API on 2026-12-11 (announced 2026-06-11), o3-deep-research shuts down 2026-07-23, and o3-mini shuts down 2026-10-23. Migration targets are GPT-5.5 (o3, o3-mini) and GPT-5.5-pro (o3-pro, o3-deep-research). Historically OpenAI's most advanced reasoning model of its era, with exceptional performance on complex coding and mathematical tasks.", + "tags": [ + "deprecated", + "reasoning", + "coding", + "mathematics", + "research", + "chain-of-thought", + "premium" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 96, + "security": 86, + "privacy_compliance": 84, + "trust_transparency": 85, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "openai-o4-mini", + "type": "model", + "name": "OpenAI o4-mini", + "provider": "OpenAI", + "description": "DEPRECATED: removed from ChatGPT 2026-02-13; o4-mini API shuts down 2026-10-23 (o4-mini-deep-research shuts down 2026-07-23); OpenAI's recommended replacement is gpt-5.4-mini (gpt-5.5-pro for deep-research). Historically OpenAI's best small reasoning model (April 2025): 93% AIME, 68% SWE-bench, first mini with full tool support + multimodality.", + "tags": [ + "deprecated", + "reasoning", + "code-generation", + "mini-model", + "budget-friendly", + "chain-of-thought", + "efficient", + "soc-2-certified" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 89, + "security": 86, + "privacy_compliance": 84, + "trust_transparency": 87, + "operational_excellence": 87 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "qwen2-5-vl-32b", + "type": "model", + "name": "Qwen2.5-VL-32B", + "provider": "Alibaba", + "description": "Multimodal vision-language model from Alibaba, now three generations behind: superseded by Qwen3-VL (Sep 2025), the natively-multimodal Qwen3.5 (released 2026-02-16), and the multimodal-input Qwen3.6 open models (Apr 2026). Historically achieved 42.9% on SWE-bench with strong image understanding at competitive pricing; new deployments should evaluate Qwen3.5/Qwen3.6 instead.", + "tags": [ + "superseded", + "vision", + "multimodal", + "open-source", + "cost-effective", + "visual-ai", + "chinese-provider", + "education", + "apache-2.0" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 86, + "security": 80, + "privacy_compliance": 79, + "trust_transparency": 82, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "qwen3-5", + "type": "model", + "name": "Qwen3.5", + "provider": "Alibaba", + "description": "Alibaba's Apache-2.0 flagship open model: Qwen3.5-397B-A17B, a 512-expert hybrid MoE (397B total / 17B active), natively multimodal, 262K context (1M on hosted Qwen3.5-Plus), 201 languages; beats Alibaba's API-only 1T Qwen3-Max with up to 19x faster long-context decode. Still its largest open-weight model as of July 2026, but smaller Apache-2.0 Qwen3.6 models (Apr 2026) surpass it on agentic coding, and the newest frontier (Qwen3.7-Max, May 2026; Qwen3.7-Plus, Jun 2026) is API-only.", + "tags": [ + "open-source", + "apache-2-0", + "multimodal", + "multilingual", + "mixture-of-experts", + "long-context", + "agentic", + "chinese-provider", + "flagship" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 92, + "security": 84, + "privacy_compliance": 79, + "trust_transparency": 82, + "operational_excellence": 87 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "qwen3-6", + "type": "model", + "name": "Qwen3.6", + "provider": "Alibaba", + "description": "Alibaba's Apache-2.0 open-weight Qwen3.6 family (Apr 2026): Qwen3.6-35B-A3B MoE (35B total / 3B active, 2026-04-16) and Qwen3.6-27B dense (2026-04-22). Hybrid Gated DeltaNet + Gated Attention with thinking mode and Thinking Preservation; 262K native context (~1M via YaRN); text, image, and video input across 201 languages. The 27B beats the 397B-A17B Qwen3.5 flagship on agentic coding (77.2% SWE-bench Verified). The newer Qwen3.7-Max/Plus frontier remains API-only.", + "tags": [ + "open-source", + "apache-2-0", + "multimodal", + "multilingual", + "agentic", + "coding", + "long-context", + "chinese-provider", + "efficient" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 91, + "security": 83, + "privacy_compliance": 79, + "trust_transparency": 81, + "operational_excellence": 86 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "activepieces", + "type": "agent", + "name": "Activepieces", + "provider": "Activepieces", + "description": "Open-source no-code business automation platform with AI agent capabilities. Self-hostable alternative to Zapier with visual workflow builder, 700+ app integrations (each also exposed as an MCP server for AI agents/LLM tools), and LLM integration for intelligent automation.", + "tags": [ + "automation", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 78, + "security": 77, + "privacy_compliance": 82, + "trust_transparency": 82, + "operational_excellence": 75 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "adala", + "type": "agent", + "name": "Adala", + "provider": "HumanSignal", + "description": "Autonomous data labeling agent framework for creating self-improving AI systems. Combines LLMs with ground truth learning to automate and improve data annotation tasks, enabling continuous learning loops.", + "tags": [ + "data-labeling", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 77, + "security": 73, + "privacy_compliance": 76, + "trust_transparency": 79, + "operational_excellence": 75 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "agentgpt", + "type": "agent", + "name": "AgentGPT", + "provider": "Reworkd", + "description": "DISCONTINUED: the AgentGPT repository was archived on 2026-01-28 (last release v1.0.0, Nov 2023) and the hosted site is frozen. Formerly a browser-based autonomous AI agent platform that let users create goal-oriented agents which break down objectives and execute tasks without continuous human intervention. Not recommended for new use.", + "tags": [ + "autonomous", + "web-based", + "archived" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 71, + "dimensions": { + "performance_reliability": 74, + "security": 68, + "privacy_compliance": 71, + "trust_transparency": 76, + "operational_excellence": 65 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "amazon-bedrock-agents", + "type": "agent", + "name": "Amazon Bedrock Agents", + "provider": "Amazon Web Services", + "description": "Fully managed AWS service for building and deploying generative AI agents with orchestration, memory, knowledge bases, and action groups. Note: AWS's strategic agent runtime is now Bedrock AgentCore (GA 2025-10-13; framework-agnostic and model-agnostic - works with models in or outside Bedrock including OpenAI and Gemini; 8-hour sessions, session isolation; expanded at re:Invent 2025) paired with the open-source Strands Agents SDK; evaluate AgentCore for new builds.", + "tags": [ + "aws" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 90, + "dimensions": { + "performance_reliability": 88, + "security": 94, + "privacy_compliance": 93, + "trust_transparency": 85, + "operational_excellence": 92 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "amazon-kiro", + "type": "agent", + "name": "Kiro", + "provider": "Amazon Web Services (AWS)", + "description": "AWS's spec-driven agentic IDE and CLI: it turns prompts into structured specs (requirements.md in EARS notation, design.md, tasks.md) before implementing, alongside a freeform vibe mode, agent hooks, and steering files. Preview 2025-07-14, GA 2025-11-17; official successor to Amazon Q Developer. Available in AWS GovCloud (US) with IAM Identity Center integration, signaling a regulated-workload posture (FedRAMP High authorization in progress, not yet granted).", + "tags": [ + "ide", + "spec-driven", + "aws", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 68, + "dimensions": { + "performance_reliability": 75, + "security": 63, + "privacy_compliance": 62, + "trust_transparency": 67, + "operational_excellence": 72 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "amazon-lex", + "type": "agent", + "name": "Amazon Lex", + "provider": "Amazon Web Services", + "description": "AWS managed conversational AI service for building chatbots and voice assistants with automatic speech recognition (ASR) and natural language understanding (NLU). Integrates natively with AWS services and enterprise systems. Lex V2 remains an active service (V1 was discontinued September 15, 2025), but AWS's strategic direction for generative/LLM-based agents is Amazon Bedrock Agents and Bedrock AgentCore; Lex is positioned for intent-based bots and Amazon Connect contact-center flows.", + "tags": [ + "aws" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 88, + "security": 92, + "privacy_compliance": 91, + "trust_transparency": 86, + "operational_excellence": 90 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "autogen", + "type": "agent", + "name": "Microsoft AutoGen", + "provider": "Microsoft Research", + "description": "MAINTENANCE MODE: AutoGen now receives bug/security fixes only, is community-managed going forward, and is superseded by the Microsoft Agent Framework (1.0 GA on 2026-04-03), the recommended migration path. AutoGen is a multi-agent conversation framework for LLM applications with conversable agents combining LLMs, human input, and tools across complex workflows.", + "tags": [ + "multi-agent", + "microsoft", + "open-source", + "maintenance-mode" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 86, + "security": 83, + "privacy_compliance": 84, + "trust_transparency": 88, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "autogpt", + "type": "agent", + "name": "AutoGPT", + "provider": "Significant Gravitas", + "description": "Autonomous AI agent project that pioneered the autonomous agent paradigm. Development has shifted from the classic self-hosted agent to the AutoGPT Platform (still in beta as of mid-2026), a visual agent builder with workflow automation, an expanding plugin/block ecosystem, and cloud-hosted deployment alongside self-hosting. The repository remains one of the most starred open-source projects (185k+ stars) with active development.", + "tags": [ + "autonomous", + "experimental", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 68, + "dimensions": { + "performance_reliability": 68, + "security": 62, + "privacy_compliance": 70, + "trust_transparency": 76, + "operational_excellence": 65 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "azure-bot-service", + "type": "agent", + "name": "Azure Bot Service", + "provider": "Microsoft", + "description": "LEGACY/RETIRING: the underlying Bot Framework SDK is retired — final long-term support ended December 31, 2025 (no further updates; Azure portal support tickets no longer serviced), and new multi-tenant bot creation was deprecated after July 31, 2025. Microsoft directs new projects to Copilot Studio or the Microsoft 365 Agents SDK (GA; C#, JavaScript, Python). Existing bots continue to function, and Azure Bot Service channels remain in use (including by Copilot Studio).", + "tags": [ + "microsoft", + "azure", + "legacy" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 88, + "dimensions": { + "performance_reliability": 87, + "security": 91, + "privacy_compliance": 90, + "trust_transparency": 85, + "operational_excellence": 89 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "babyagi", + "type": "agent", + "name": "BabyAGI", + "provider": "Yohei Nakajima", + "description": "ARCHIVED: the original BabyAGI repo was archived to babyagi_archive in September 2024 and replaced by an experimental self-building framework; it is not production-maintained. Originally a minimalist autonomous task-driven AI agent that created, prioritized, and executed tasks toward an objective, demonstrating AGI concepts in under 200 lines of code.", + "tags": [ + "autonomous", + "experimental", + "open-source", + "archived" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 66, + "dimensions": { + "performance_reliability": 64, + "security": 58, + "privacy_compliance": 67, + "trust_transparency": 82, + "operational_excellence": 57 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "bytedance-trae", + "type": "agent", + "name": "Trae", + "provider": "ByteDance", + "description": "ByteDance's free AI IDE (VS Code fork) with Builder agent mode and the autonomous SOLO agent (standalone app since March 2026). Aggressive free access to frontier models (Claude, GPT, DeepSeek) drove rapid adoption after its early-2025 launch. Its trust record is the story: 2025 analyses found telemetry continuing after opt-out (~500 network calls in ~7 minutes), persistent hardware-derived device IDs, 5-year post-account data retention, and ByteDance jurisdiction concerns.", + "tags": [ + "ide", + "agentic-coding", + "privacy-risk", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 50, + "dimensions": { + "performance_reliability": 69, + "security": 39, + "privacy_compliance": 24, + "trust_transparency": 53, + "operational_excellence": 65 + }, + "strengths_count": 5, + "limitations_count": 7 + }, + { + "id": "chatgpt-agent", + "type": "agent", + "name": "ChatGPT Agent Mode", + "provider": "OpenAI", + "description": "Agent mode in ChatGPT that gives the assistant its own virtual computer, merging Operator's web browsing with deep research's analysis. Launched 2025-07-17; Agent Mode v2 (2026) adds persistent memory, scheduled tasks, and GitHub/Jira connectors. Runs in an isolated cloud VM with watch mode for sensitive sites and confirmations before consequential actions, while OpenAI openly acknowledges prompt injection as a core unsolved risk. Included in Plus, Pro, and Team plans with usage caps.", + "tags": [ + "autonomous", + "general-purpose", + "cloud-agent", + "consumer", + "openai" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 69, + "dimensions": { + "performance_reliability": 75, + "security": 67, + "privacy_compliance": 60, + "trust_transparency": 68, + "operational_excellence": 76 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "claude-agent-sdk", + "type": "agent", + "name": "Claude Agent SDK", + "provider": "Anthropic", + "description": "SDK exposing Claude Code's production agent harness (tool loop, permission system, subagents, MCP) for building general-purpose agents in TypeScript and Python. Renamed from Claude Code SDK in September 2025.", + "tags": [ + "agent-sdk", + "anthropic", + "framework" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 86, + "security": 78, + "privacy_compliance": 74, + "trust_transparency": 78, + "operational_excellence": 80 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "claude-code", + "type": "agent", + "name": "Claude Code", + "provider": "Anthropic", + "description": "Anthropic's agentic coding tool available as a terminal CLI, IDE extensions, web, and desktop app. Plans and executes multi-step coding tasks with tiered permissions, OS-level sandboxing, MCP integration, hooks, subagents, and plugins/skills.", + "tags": [ + "coding-agent", + "cli", + "anthropic", + "sandboxed" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 88, + "security": 77, + "privacy_compliance": 71, + "trust_transparency": 80, + "operational_excellence": 84 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "claude-cowork", + "type": "agent", + "name": "Claude Cowork", + "provider": "Anthropic", + "description": "Anthropic's agentic workspace for non-technical knowledge work — 'Claude Code for the office.' Launched 2026-01-12 as a macOS desktop app that runs tasks in a sandboxed local VM (Apple Virtualization Framework with a custom Linux rootfs), expanding to web and mobile with cloud execution in July 2026. Turns natural-language goals into reports, spreadsheets, and organized files. Shipped with a disclosed prompt-injection-via-malicious-files risk that remains its central security tension.", + "tags": [ + "autonomous", + "knowledge-work", + "consumer", + "sandboxed", + "anthropic" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 73, + "dimensions": { + "performance_reliability": 83, + "security": 65, + "privacy_compliance": 68, + "trust_transparency": 73, + "operational_excellence": 75 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "cline", + "type": "agent", + "name": "Cline", + "provider": "Cline Bot Inc.", + "description": "The most-adopted open-source coding agent (5M+ installs by Feb 2026), formerly 'Claude Dev'. Apache-2.0, runs in VS Code, JetBrains, terminal CLI, and via an SDK. Client-side BYOK architecture keeps code local while routing to any model provider; Plan/Act modes with human approval gate edits and commands. Security researchers demonstrated prompt-injection paths that bypassed command approval (mitigated in v3.35.0), and Anthropic's 2026 OAuth crackdown cut off Claude subscription use.", + "tags": [ + "coding-agent", + "open-source", + "byok", + "ide-extension", + "mcp" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 78, + "dimensions": { + "performance_reliability": 79, + "security": 68, + "privacy_compliance": 83, + "trust_transparency": 85, + "operational_excellence": 76 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "crewai", + "type": "agent", + "name": "CrewAI", + "provider": "CrewAI Inc.", + "description": "Role-playing multi-agent framework for orchestrating collaborative autonomous agents. Agents work as a crew with defined roles, goals, and backstories to tackle complex tasks through delegation. Commercial offerings: the CrewAI AMP managed platform and self-hosted CrewAI Factory. Note: four Code Interpreter CVEs (CVE-2026-2275/2285/2286/2287) disclosed 2026-03-30; vendor reports all fixed in current releases, with the built-in CodeInterpreterTool removed in favor of external sandboxes.", + "tags": [ + "multi-agent", + "collaborative", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 81, + "security": 72, + "privacy_compliance": 80, + "trust_transparency": 83, + "operational_excellence": 79 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "cursor-agent", + "type": "agent", + "name": "Cursor Agent", + "provider": "Anysphere (SpaceX acquisition announced 2026-06-16, expected to close Q3 2026)", + "description": "Agent mode of Cursor, Anysphere's AI-native IDE. The product's primary surface is now agentic: parallel local agents, cloud/background agents running in isolated VMs, and the in-house Composer model line (Composer 2.5) alongside Claude, GPT, and Gemini. Anysphere IPO'd on Nasdaq in June 2026 and days later agreed to a $60B all-stock acquisition by SpaceX (announced 2026-06-16, expected to close Q3 2026 under its xAI subsidiary).", + "tags": [ + "ide", + "agentic-coding", + "parallel-agents", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 72, + "dimensions": { + "performance_reliability": 83, + "security": 66, + "privacy_compliance": 64, + "trust_transparency": 69, + "operational_excellence": 79 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "devin", + "type": "agent", + "name": "Devin", + "provider": "Cognition", + "description": "Autonomous AI software engineer from Cognition that plans and executes multi-step engineering tasks in a sandboxed cloud workspace with its own editor, shell, and browser, and delivers work as pull requests. Now spans Devin Cloud agents, Devin Desktop (the rebranded Windsurf IDE, June 2026) with the Rust-based Devin Local agent, and Cognition's in-house SWE model line (SWE-1.7, July 2026).", + "tags": [ + "autonomous", + "software-engineering", + "cloud-agent", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 71, + "dimensions": { + "performance_reliability": 80, + "security": 69, + "privacy_compliance": 61, + "trust_transparency": 66, + "operational_excellence": 77 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "dify", + "type": "agent", + "name": "Dify", + "provider": "LangGenius", + "description": "Open-source LLM application and agentic workflow platform with a visual canvas, built-in RAG pipeline, agent nodes, and 50+ tools. One of the most-starred LLM app platforms (148k+ stars), self-hosted or via Dify Cloud. Moderate-severity 2026 advisories (authorization bypass CVE-2026-41949, path traversal CVE-2026-41948, XSS CVE-2026-6619, account enumeration CVE-2026-28288) make staying on current releases (1.15.0+) important; no critical RCE or in-the-wild exploitation reported.", + "tags": [ + "no-code", + "workflow-platform", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 80, + "security": 76, + "privacy_compliance": 83, + "trust_transparency": 85, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "e2b-agents", + "type": "agent", + "name": "E2B Agents", + "provider": "E2B", + "description": "Secure cloud runtime for AI agents with code interpreter capabilities. Provides sandboxed environments for executing agent-generated code safely, with support for multiple programming languages and pre-built integrations.", + "tags": [ + "sandbox", + "code-execution" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 81, + "security": 89, + "privacy_compliance": 80, + "trust_transparency": 83, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "factory-droids", + "type": "agent", + "name": "Factory Droids", + "provider": "Factory AI", + "description": "Factory AI's enterprise agent-native software development platform. Specialist autonomous Droids handle coding, testing, code review, refactoring, and DevOps work across Desktop, CLI, SDK, Slack, ticketing, and CI, with Missions enabling long-horizon multi-agent workflows. #1 on Terminal-Bench (Sept 2025); $220M raised at a $1.5B valuation (Apr 2026); used daily by hundreds of thousands of developers at enterprises including Nvidia, Adobe, and EY.", + "tags": [ + "autonomous", + "enterprise", + "sdlc-platform", + "multi-agent", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 74, + "dimensions": { + "performance_reliability": 82, + "security": 69, + "privacy_compliance": 74, + "trust_transparency": 66, + "operational_excellence": 81 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "flowise", + "type": "agent", + "name": "Flowise", + "provider": "FlowiseAI (Workday since Aug 2025)", + "description": "Open-source low-code platform for LLM orchestration flows and AI agents, acquired by Workday August 2025. Visual node editor for RAG pipelines, chatbots, agents. SECURITY: three CVEs with confirmed in-the-wild exploitation - CVE-2025-59528 (CVSS 10.0 RCE via CustomMCP node, fixed in 3.0.6, exploited from April 2026, 12,000+ instances exposed), CVE-2025-8943 (CVSS 9.8 OS command RCE), CVE-2025-26319 (arbitrary file upload). Upgrade to 3.1.x; do not expose unauthenticated instances.", + "tags": [ + "visual", + "low-code", + "open-source", + "security-incidents" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 78, + "security": 68, + "privacy_compliance": 79, + "trust_transparency": 84, + "operational_excellence": 76 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "gemini-cli", + "type": "agent", + "name": "Gemini CLI", + "provider": "Google", + "description": "Open-source terminal AI agent from Google that brings Gemini into the command line. Uses a ReAct loop with built-in tools, MCP server support, and Google Search grounding. Consumer/individual access (free tier and Google AI Pro/Ultra) ended 2026-06-18 as Google transitioned individuals to the closed-source Antigravity CLI ('agy'); Gemini CLI remains available to Gemini Code Assist Standard/Enterprise organizations and paid Gemini API key users.", + "tags": [ + "cli", + "open-source", + "coding-agent", + "google" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 77, + "security": 80, + "privacy_compliance": 69, + "trust_transparency": 84, + "operational_excellence": 70 + }, + "strengths_count": 4, + "limitations_count": 5 + }, + { + "id": "github-copilot-coding-agent", + "type": "agent", + "name": "GitHub Copilot Coding Agent", + "provider": "GitHub (Microsoft)", + "description": "Autonomous background coding agent built into GitHub. Assign it a GitHub issue or prompt and it works in an ephemeral GitHub Actions sandbox, then opens a draft pull request for human review. Distinct from Copilot's interactive IDE agent mode.", + "tags": [ + "coding-agent", + "autonomous", + "github", + "enterprise" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 74, + "dimensions": { + "performance_reliability": 76, + "security": 74, + "privacy_compliance": 65, + "trust_transparency": 72, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "glean-ai", + "type": "agent", + "name": "Glean AI", + "provider": "Glean Technologies Inc.", + "description": "Enterprise 'Work AI' platform that unifies information across business tools and applications. Has expanded well beyond enterprise search into AI assistants and autonomous agents that execute tasks across the business (Glean Assistant, Glean Agents, plus Glean Protect for agent governance/security). Raised a $150M Series F at a $7.2B valuation (June 2025) and crossed $300M ARR by May 2026; customers include Dell, Workday, and Palo Alto Networks.", + "tags": [ + "enterprise", + "knowledge-management", + "search", + "productivity" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 88, + "dimensions": { + "performance_reliability": 88, + "security": 91, + "privacy_compliance": 90, + "trust_transparency": 85, + "operational_excellence": 87 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "google-adk", + "type": "agent", + "name": "Google Agent Development Kit (ADK)", + "provider": "Google", + "description": "Open-source, code-first framework for building, evaluating, and deploying AI agents. Supports workflow agents, multi-agent hierarchies, built-in evaluation, and deployment to Vertex AI Agent Engine. Underlies Google's broader agent stack and is Gemini-optimized but model-agnostic.", + "tags": [ + "multi-agent", + "open-source", + "google", + "framework" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 85, + "security": 79, + "privacy_compliance": 82, + "trust_transparency": 86, + "operational_excellence": 84 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "google-agent-builder", + "type": "agent", + "name": "Gemini Enterprise Agent Platform (formerly Vertex AI Agent Builder)", + "provider": "Google Cloud", + "description": "REBRANDED: at Cloud Next (April 2026) Vertex AI Agent Builder became the Gemini Enterprise Agent Platform (APIs unchanged), following Agentspace's absorption into Gemini Enterprise in Oct 2025. Console migration completed ~May 21, 2026: Vertex AI branding is gone, with all Vertex AI capabilities under the new name and no breaking changes. Google Cloud's managed platform for conversational AI agents and search apps with no-code/low-code options, enterprise search, and grounding.", + "tags": [ + "google", + "rebranded" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 87, + "security": 92, + "privacy_compliance": 91, + "trust_transparency": 84, + "operational_excellence": 90 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "google-antigravity", + "type": "agent", + "name": "Google Antigravity", + "provider": "Google", + "description": "Google's agent-first development platform (launched 2025-11-18 with Gemini 3): an IDE built on licensed Windsurf code where an agent manager orchestrates autonomous agents across editor, terminal, and browser, verifying work through artifacts (plans, task lists, screenshots, recordings). Also the home of the closed-source 'agy' CLI that replaced consumer Gemini CLI (June 2026). Free public preview whose first months were marred by serious, well-documented security failures.", + "tags": [ + "ide", + "agentic-coding", + "parallel-agents", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 59, + "dimensions": { + "performance_reliability": 76, + "security": 42, + "privacy_compliance": 52, + "trust_transparency": 61, + "operational_excellence": 63 + }, + "strengths_count": 5, + "limitations_count": 7 + }, + { + "id": "google-dialogflow", + "type": "agent", + "name": "Google Conversational Agents (Dialogflow CX)", + "provider": "Google Cloud", + "description": "REBRANDED: Dialogflow CX is now sold as Conversational Agents on Google Cloud - the standalone CX console was deprecated October 31, 2025, and users are auto-routed to the Conversational Agents console, where generative playbooks and LLM-driven fulfillment sit alongside classic deterministic flows. Google's advanced conversational AI platform for building virtual agents with visual flow design, state management, and enterprise features. Successor to Dialogflow ES.", + "tags": [ + "google", + "rebranded" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 88, + "dimensions": { + "performance_reliability": 89, + "security": 90, + "privacy_compliance": 88, + "trust_transparency": 87, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "google-jules", + "type": "agent", + "name": "Google Jules", + "provider": "Google", + "description": "Asynchronous autonomous coding agent from Google. Jules clones a repository into an isolated Google Cloud VM, plans and writes code in the background, runs tests, and opens pull requests for human review. Powered by Gemini 2.5 Pro on the free tier and Gemini 3 Pro on paid Google AI Pro/Ultra tiers.", + "tags": [ + "coding-agent", + "autonomous", + "asynchronous", + "google" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 71, + "dimensions": { + "performance_reliability": 76, + "security": 71, + "privacy_compliance": 58, + "trust_transparency": 69, + "operational_excellence": 79 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "goose", + "type": "agent", + "name": "Goose", + "provider": "Agentic AI Foundation (Linux Foundation); originally Block", + "description": "Open-source, local-first AI agent written in Rust, created by Block and donated to the Linux Foundation's Agentic AI Foundation (repo moved to aaif-goose/goose, April 2026). Ships as a CLI and desktop app, is MCP-native with 70+ extensions, and works with 15+ LLM providers including fully local models via Ollama. Free under Apache-2.0; vendor-neutral foundation governance is its core trust story, with the usual MCP/prompt-injection surface of local agents.", + "tags": [ + "coding-agent", + "open-source", + "local-first", + "mcp", + "foundation-governed" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 78, + "dimensions": { + "performance_reliability": 77, + "security": 69, + "privacy_compliance": 85, + "trust_transparency": 83, + "operational_excellence": 77 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "haystack", + "type": "agent", + "name": "Haystack", + "provider": "deepset", + "description": "Open-source AI orchestration framework from deepset for building production-ready LLM applications, RAG pipelines, agent workflows, and semantic search systems. Modular architecture with pre-built components for document processing, retrieval, and generation, plus Agent components added in the 2.x line. Actively maintained (2.31.0 released July 2026) with commercial support via Haystack Enterprise and the deepset AI Platform.", + "tags": [ + "rag", + "search", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 84, + "security": 76, + "privacy_compliance": 83, + "trust_transparency": 88, + "operational_excellence": 80 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "ibm-watson-assistant", + "type": "agent", + "name": "IBM watsonx Assistant (now part of watsonx Orchestrate)", + "provider": "IBM", + "description": "CONSOLIDATED: Watson Assistant was rebranded watsonx Assistant (2023) and folded into IBM watsonx Orchestrate - the Assistant pricing page now redirects to Orchestrate, and IBM positions new implementations on Orchestrate, its agentic AI platform (next generation announced at Think 2026, May 2026, as a multi-agent orchestration and governance control plane). Existing Assistant deployments continue to run. Combines NLU, dialog skills, and integrations for enterprise virtual agents.", + "tags": [ + "ibm", + "enterprise", + "rebranded" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 86, + "security": 89, + "privacy_compliance": 88, + "trust_transparency": 84, + "operational_excellence": 87 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "jetbrains-junie", + "type": "agent", + "name": "Junie", + "provider": "JetBrains", + "description": "JetBrains' AI coding agent, out of beta since June 2026 and ranked the #1 coding agent on the independent SWE-Rebench benchmark (61.6% resolved, 72.7% pass@5). Junie plans before it codes, debugs with the IDE's real debugger, reviews pull requests with project context, and runs from JetBrains IDEs, a terminal CLI, or CI (GitHub Actions/GitLab). LLM-agnostic with BYOK and local-model support; included in JetBrains AI subscription tiers.", + "tags": [ + "coding-agent", + "ide-native", + "jetbrains", + "llm-agnostic", + "byok" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 73, + "dimensions": { + "performance_reliability": 80, + "security": 65, + "privacy_compliance": 73, + "trust_transparency": 71, + "operational_excellence": 78 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "kore-ai", + "type": "agent", + "name": "Kore.ai", + "provider": "Kore.ai Inc.", + "description": "Enterprise agentic AI platform for designing, deploying, managing, and scaling AI agents across business operations. In May 2026 Kore.ai launched the Artemis edition of its Agent Platform, built around the Agent Blueprint Language (ABL), a compiled, declarative YAML-based language for defining, validating, and governing agents and multi-agent systems, initially on Microsoft Azure. Secured a strategic growth investment led by AllianceBernstein in January 2026 (total funding ~$620M).", + "tags": [ + "enterprise", + "conversational-ai", + "customer-support", + "no-code" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 87, + "security": 90, + "privacy_compliance": 88, + "trust_transparency": 84, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "langflow", + "type": "agent", + "name": "Langflow", + "provider": "IBM (formerly DataStax/Logspace)", + "description": "Visual builder for LangChain-based AI apps and agents; owned by IBM via the Feb 2025 DataStax acquisition. SECURITY: recurring CVEs - CVE-2025-3248 (CVSS 9.8 unauthenticated RCE, fixed in 1.3.0, CISA KEV, Flodrix botnet), CVE-2025-34291 (account takeover/RCE), CVE-2026-33017 (second 9.8 unauthenticated RCE, exploited in 2026 for cryptomining, fixed in 1.9.0), CVE-2026-55255 (IDOR, fixed in 1.9.2), CVE-2026-5027 (path traversal). Upgrade to 1.10.1+; never expose unauthenticated.", + "tags": [ + "visual", + "low-code", + "open-source", + "security-incidents" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 79, + "security": 64, + "privacy_compliance": 77, + "trust_transparency": 85, + "operational_excellence": 78 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "langgraph-agent", + "type": "agent", + "name": "LangGraph Agent", + "provider": "LangChain", + "description": "LangChain's graph-based agent framework for building stateful, multi-actor applications with cycles and controllable execution flow. Reached 1.0 GA on 2025-10-22, adding durable execution and middleware. Enables complex, cyclic agent workflows with human-in-the-loop capabilities and production-grade persistence.", + "tags": [ + "workflow", + "langchain", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 84, + "security": 79, + "privacy_compliance": 82, + "trust_transparency": 86, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "llamaindex-agent", + "type": "agent", + "name": "LlamaIndex Agent", + "provider": "LlamaIndex", + "description": "Data framework optimized for building LLM applications with advanced RAG (Retrieval-Augmented Generation) capabilities. Agents can reason over complex data sources using sophisticated query engines and retrieval strategies; the event-driven Workflows engine (llama-index-workflows 2.x) has matured agent orchestration well beyond the earlier ReAct-only story. Managed cloud offerings (LlamaCloud parse/extract/index, LlamaAgents) are available alongside the MIT-licensed framework.", + "tags": [ + "rag", + "retrieval", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 83, + "security": 77, + "privacy_compliance": 81, + "trust_transparency": 87, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "lovable", + "type": "agent", + "name": "Lovable", + "provider": "Lovable AB", + "description": "Leading prompt-to-app 'vibe coding' platform from Stockholm-based Lovable AB (ex GPT Engineer): natural language in, deployed full-stack web app out, with Supabase-backed Lovable Cloud. Launched Nov 2024; crossed $400M ARR in Feb 2026 with ~8M users ($330M Series B at $6.6B, Dec 2025). The epicenter of the vibe-coding security debate: its own posture is certified (SOC 2, ISO 27001), but apps it generates have repeatedly shipped without proper access controls.", + "tags": [ + "app-builder", + "vibe-coding", + "no-code", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 61, + "dimensions": { + "performance_reliability": 67, + "security": 51, + "privacy_compliance": 54, + "trust_transparency": 62, + "operational_excellence": 70 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "make-ai", + "type": "agent", + "name": "Make AI", + "provider": "Make (Integromat)", + "description": "Visual automation platform with AI capabilities for building no-code intelligent workflows. Integrates AI models with 3,000+ app connections for creating automated business processes, and offers Make AI Agents (available on all paid plans) that can reason, decide next steps, and trigger workflows.", + "tags": [ + "automation", + "workflow" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 86, + "security": 88, + "privacy_compliance": 87, + "trust_transparency": 85, + "operational_excellence": 88 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "manus", + "type": "agent", + "name": "Manus", + "provider": "Butterfly Effect / Meta (acquisition being unwound as of June 2026)", + "description": "General-purpose autonomous agent that executes end-to-end tasks in a cloud VM equipped with a browser, shell, and file tools. Launched virally in March 2025 by Butterfly Effect and acquired by Meta for over $2B in late December 2025 — but Chinese regulators ordered the deal reversed in April 2026, and by June 2026 Meta had completed an operational split (halting data sharing) while it dismantles the acquisition; Manus co-founders are reportedly in talks to buy the company back.", + "tags": [ + "autonomous", + "general-purpose", + "cloud-agent", + "consumer" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 64, + "dimensions": { + "performance_reliability": 76, + "security": 60, + "privacy_compliance": 50, + "trust_transparency": 64, + "operational_excellence": 68 + }, + "strengths_count": 5, + "limitations_count": 6 + }, + { + "id": "mastra", + "type": "agent", + "name": "Mastra", + "provider": "Mastra AI (YC W25)", + "description": "TypeScript-first AI agent framework from the Gatsby founders, combining agents, durable workflows, RAG, and evals in one toolkit. Provider-agnostic via Vercel AI SDK model routing, with a local dev playground and 1.0 stable release in January 2026.", + "tags": [ + "typescript", + "workflows", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 81, + "dimensions": { + "performance_reliability": 83, + "security": 72, + "privacy_compliance": 82, + "trust_transparency": 85, + "operational_excellence": 84 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "memgpt", + "type": "agent", + "name": "Letta (formerly MemGPT)", + "provider": "Letta Inc.", + "description": "REBRANDED: MemGPT became Letta (Letta Inc., a UC Berkeley spinout) in September 2024; \"MemGPT\" now refers only to the research technique. Letta is a memory-enhanced LLM agent platform enabling long-term context via OS-inspired virtual context management. In March 2026 Letta announced a pivot to Letta Code, a client-side, model-agnostic agent harness with persistent memory, deprecating server-side memory tools, templates, identities, MCP integrations, and tool rules.", + "tags": [ + "memory", + "long-context", + "open-source", + "rebranded" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 78, + "dimensions": { + "performance_reliability": 79, + "security": 75, + "privacy_compliance": 78, + "trust_transparency": 82, + "operational_excellence": 76 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "microsoft-agent-framework", + "type": "agent", + "name": "Microsoft Agent Framework", + "provider": "Microsoft", + "description": "Open-source SDK and runtime for building AI agents and graph-based multi-agent workflows in .NET and Python. Merges AutoGen and Semantic Kernel into a single framework with checkpointing, middleware, and OpenTelemetry-based observability.", + "tags": [ + "multi-agent", + "open-source", + "enterprise", + "workflow-orchestration" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 85, + "security": 80, + "privacy_compliance": 84, + "trust_transparency": 87, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "microsoft-scout", + "type": "agent", + "name": "Microsoft Scout", + "provider": "Microsoft", + "description": "Always-on autonomous personal agent for Microsoft 365, built on the open-source OpenClaw runtime. Announced at Build 2026 (2026-06-02) as an experimental Frontier-program release: a Windows/macOS desktop app that reads/writes files, runs shell commands, drives a browser, and manages email, calendar, and Teams. Pairs 2026's strongest enterprise governance — per-agent Entra identity, task-scoped credentials, in-line Purview DLP, human sign-off — with a heavily CVE-burdened OSS core.", + "tags": [ + "autonomous", + "always-on", + "enterprise", + "microsoft", + "openclaw-based" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 70, + "dimensions": { + "performance_reliability": 75, + "security": 64, + "privacy_compliance": 74, + "trust_transparency": 74, + "operational_excellence": 63 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "n8n-ai-agent", + "type": "agent", + "name": "n8n AI Agent", + "provider": "n8n", + "description": "Fair-code workflow automation platform with AI agent capabilities, now on the 2.x major version. Visual builder with AI agent nodes and 400+ integrations. Valued at $5.2B (May 2026) after a $180M Series C (Oct 2025) and a strategic SAP investment. SECURITY: critical CVEs disclosed Jan-Feb 2026 - CVE-2026-21858 'Ni8mare' (unauthenticated RCE), CVE-2026-21877 (CVSS 9.9 RCE), CVE-2026-25049 (authenticated RCE) - all patched; n8n 2.5.2+ is the minimum safe self-hosted version.", + "tags": [ + "automation", + "workflow" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 83, + "security": 77, + "privacy_compliance": 84, + "trust_transparency": 87, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "openai-agents-sdk", + "type": "agent", + "name": "OpenAI Agents SDK", + "provider": "OpenAI", + "description": "Production-ready multi-agent orchestration framework built around agents, handoffs, guardrails, and tracing. Open-source (MIT) successor to Swarm, released March 2025 with a major overhaul in April 2026 that added a model-native harness (filesystem tools, shell execution, apply-patch edits) and native sandboxed execution for long-horizon tasks.", + "tags": [ + "multi-agent", + "open-source", + "orchestration", + "openai" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 84, + "security": 79, + "privacy_compliance": 80, + "trust_transparency": 90, + "operational_excellence": 87 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "openai-assistants-api", + "type": "agent", + "name": "OpenAI Assistants API", + "provider": "OpenAI", + "description": "DEPRECATED: the Assistants API will be sunset on 2026-08-26 (under 7 weeks away); after that date all /v1/assistants, /v1/threads, and /v1/threads/runs calls return errors. OpenAI directs users to the Responses API plus Conversations API as the replacement; Azure OpenAI Assistants retires the same day. Previously a managed agent framework with native tool use, code interpreter, file search, and persistent threads. Do not start new projects on it.", + "tags": [ + "openai", + "assistants", + "deprecated" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 85, + "dimensions": { + "performance_reliability": 89, + "security": 87, + "privacy_compliance": 86, + "trust_transparency": 83, + "operational_excellence": 78 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "openai-codex", + "type": "agent", + "name": "OpenAI Codex", + "provider": "OpenAI", + "description": "OpenAI's coding agent spanning a cloud agent that runs tasks in isolated containers and an open-source CLI. Delegates parallel software tasks (features, fixes, PRs) powered by GPT-5.5 (recommended model as of mid-2026; GPT-5.3-Codex deprecated), with network access disabled by default in the cloud.", + "tags": [ + "coding-agent", + "cloud-sandbox", + "openai", + "parallel-tasks" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 84, + "security": 84, + "privacy_compliance": 73, + "trust_transparency": 84, + "operational_excellence": 84 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "openclaw", + "type": "agent", + "name": "OpenClaw", + "provider": "OpenClaw Foundation", + "description": "Viral MIT-licensed open-source personal AI agent (formerly Clawdbot, then Moltbot) stewarded by the OpenClaw Foundation with OpenAI backing. Runs on the user's own machine and acts through WhatsApp, Telegram, Signal, Discord, and iMessage; it browses, emails, shops, and controls the desktop with any BYO LLM. The fastest repo to ~350K+ GitHub stars and 2026's defining agent-security story: hundreds of CVEs, mass-exposed instances, and malicious ClawHub skills.", + "tags": [ + "personal-agent", + "open-source", + "messaging", + "self-hosted", + "high-risk" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 60, + "dimensions": { + "performance_reliability": 75, + "security": 37, + "privacy_compliance": 56, + "trust_transparency": 78, + "operational_excellence": 55 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "opencode", + "type": "agent", + "name": "OpenCode", + "provider": "SST / Anomaly Innovations", + "description": "MIT-licensed open-source terminal AI coding agent from the SST team (Anomaly Innovations). TUI-first with a client/server architecture, 75+ model providers, an optional curated Zen gateway, and millions of monthly developers. GitHub made Copilot subscriptions work with it (Jan 2026), while Anthropic blocked its consumer-OAuth access the same month, forcing removal of Claude Pro/Max support - a live case study in agent supply-chain and ToS risk.", + "tags": [ + "coding-agent", + "terminal", + "open-source", + "byok", + "multi-provider" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 78, + "security": 67, + "privacy_compliance": 80, + "trust_transparency": 85, + "operational_excellence": 76 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "perplexity-comet", + "type": "agent", + "name": "Comet", + "provider": "Perplexity AI", + "description": "Perplexity's Chromium-based agentic browser. Comet Assistant runs in a sidebar with full context of open tabs and can fill forms, book travel, buy products, and manage email and calendars on the user's behalf. Launched on desktop 2025-07-09, Android 2025-11-20, and iOS 2026-03-18, with Comet Enterprise arriving March 2026. Comet is the canonical browser-agent risk surface: the 'CometJacking' prompt-injection class (2025) was demonstrated against it.", + "tags": [ + "agentic-browser", + "consumer", + "prompt-injection-risk", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 63, + "dimensions": { + "performance_reliability": 73, + "security": 49, + "privacy_compliance": 55, + "trust_transparency": 62, + "operational_excellence": 75 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "poke", + "type": "agent", + "name": "Poke", + "provider": "The Interaction Company of California", + "description": "Consumer AI agent living entirely in messaging — iMessage/SMS, Telegram, WhatsApp — with no app to install. Routes each task to the best-fit model across providers; handles calendar, email, smart home, health, and purchases via recipes spanning 40+ integrations. Publicly launched March 2026; became the first third-party AI agent approved on Apple's Messages for Business (2026-06-04). Standing email/calendar access through a channel with no OS permission model is a novel risk surface.", + "tags": [ + "consumer", + "messaging-native", + "personal-assistant", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 55, + "dimensions": { + "performance_reliability": 71, + "security": 46, + "privacy_compliance": 44, + "trust_transparency": 51, + "operational_excellence": 61 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "pydantic-ai", + "type": "agent", + "name": "Pydantic AI", + "provider": "Pydantic", + "description": "Type-safe Python agent framework from the creators of Pydantic. Reached v1.0 stable on 2025-09-04 with a formal API stability commitment, then v2.0 on 2026-06-23 introducing a harness-first design with 'capabilities' (composable bundles of tools, hooks, instructions, and model settings) as a core primitive. Provides production-ready agents with strong typing, validation, and structured outputs, designed for reliability and maintainability in production systems.", + "tags": [ + "python", + "type-safe", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 85, + "security": 82, + "privacy_compliance": 81, + "trust_transparency": 88, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "rasa", + "type": "agent", + "name": "Rasa (Open Source & Rasa Pro)", + "provider": "Rasa Technologies", + "description": "Open-source-rooted conversational AI framework for building contextual assistants with full control over NLU, dialogue management, and deployment. NOTE: classic Rasa Open Source (Apache 2.0) is in maintenance mode; Rasa's strategic direction is CALM, its LLM-native dialogue approach, delivered via the commercially licensed Rasa Pro (3.16.x as of mid-2026) and the no-code Rasa Studio. A free Rasa Pro Developer Edition covers 1,000 conversations/month; paid tiers start at ~$35,000/year.", + "tags": [ + "nlp", + "conversational", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 82, + "security": 78, + "privacy_compliance": 87, + "trust_transparency": 86, + "operational_excellence": 77 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "relevance-ai", + "type": "agent", + "name": "Relevance AI", + "provider": "Relevance AI Pty Ltd", + "description": "No-code AI agent platform that enables teams to build, customize, and deploy AI workforce agents. Features visual workflow builders, tool integrations, and the ability to create specialized AI employees for various business functions including sales, support, and research. Raised a $24M Series B (May 2025, led by Bessemer) on the back of rapid growth (40,000 agents registered in January 2025 alone); pricing now splits usage into Actions and Vendor Credits, with BYO LLM API keys on paid plans.", + "tags": [ + "no-code", + "workflow-automation", + "sales", + "marketing", + "research" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 82, + "security": 78, + "privacy_compliance": 76, + "trust_transparency": 82, + "operational_excellence": 84 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "replit-agent", + "type": "agent", + "name": "Replit Agent", + "provider": "Replit", + "description": "Full-stack app-building agent inside Replit's browser-based cloud workspace: it plans, codes, provisions databases, and deploys from natural language. Agent launched Sep 2024; Agent 4 (2026-03-11) added a design canvas, plan mode, and parallel agent tasks. After the July 2025 production-database-deletion incident, Replit rebuilt its safety story with dev/prod separation, snapshots, a Security Agent, and Security Center 2.0. 50M+ users; $400M Series D at $9B (Mar 2026).", + "tags": [ + "app-builder", + "vibe-coding", + "cloud-agent", + "proprietary" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 68, + "dimensions": { + "performance_reliability": 76, + "security": 63, + "privacy_compliance": 57, + "trust_transparency": 67, + "operational_excellence": 76 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "salesforce-einstein-bots", + "type": "agent", + "name": "Salesforce Einstein Bots", + "provider": "Salesforce", + "description": "TRANSITIONING: Einstein Copilot was retired into Agentforce (Jan 2025, Spring '25 release) and Salesforce is steering Einstein Bots customers toward Agentforce. Einstein Bots has no announced end-of-life as of mid-2026 but is effectively legacy: Legacy Chat was fully retired February 14, 2026 (bots must run on Messaging for In-App and Web), Article Answers was retired December 31, 2025, and a 'Create AI Agents from Einstein Bots' scaffolding tool aids migration to Agentforce agents.", + "tags": [ + "salesforce", + "enterprise", + "rebranded" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 87, + "dimensions": { + "performance_reliability": 85, + "security": 90, + "privacy_compliance": 89, + "trust_transparency": 83, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "semantic-kernel-agent", + "type": "agent", + "name": "Semantic Kernel Agent", + "provider": "Microsoft", + "description": "MAINTENANCE MODE: Semantic Kernel now receives bug/security fixes only and is superseded by the Microsoft Agent Framework (1.0 GA on 2026-04-03), the recommended migration path. It remains Microsoft's enterprise SDK for integrating LLMs with conventional languages via plugins, planners, and memory, with .NET, Python, and Java support.", + "tags": [ + "microsoft", + "azure", + "open-source", + "maintenance-mode" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 89, + "dimensions": { + "performance_reliability": 87, + "security": 89, + "privacy_compliance": 90, + "trust_transparency": 90, + "operational_excellence": 91 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "sierra-ai", + "type": "agent", + "name": "Sierra", + "provider": "Sierra Technologies Inc.", + "description": "Conversational AI platform for autonomous customer experience agents, built on Sierra's Agent OS 2.0 spanning chat, voice, email, SMS, and WhatsApp. Founded by Bret Taylor (ex-Salesforce co-CEO) and Clay Bavor (ex-Google), Sierra focuses on CX agents that take real actions. One of the most heavily funded AI agent startups: $350M at a $10B valuation (Sep 2025, Greenoaks), then a $950M Series E at $15.8B (May 2026, Tiger Global and GV); 40%+ of the Fortune 50 are customers, ARR $150M+.", + "tags": [ + "customer-support", + "conversational-ai", + "autonomous-agents", + "e-commerce" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 86, + "security": 85, + "privacy_compliance": 82, + "trust_transparency": 80, + "operational_excellence": 84 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "smolagents", + "type": "agent", + "name": "smolagents", + "provider": "Hugging Face", + "description": "Minimalist Python agent library from Hugging Face. Its signature CodeAgent writes actions as executable Python code instead of JSON tool calls, enabling expressive multi-step behavior with a deliberately small core codebase.", + "tags": [ + "code-agent", + "minimalist", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 78, + "security": 66, + "privacy_compliance": 84, + "trust_transparency": 87, + "operational_excellence": 79 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "strands-agents", + "type": "agent", + "name": "Strands Agents", + "provider": "Amazon Web Services", + "description": "Open-source, model-driven AI agent SDK from AWS, used internally by Amazon Q Developer. Takes a lightweight model-first approach with MCP and A2A support, multi-agent primitives, and optional pairing with Amazon Bedrock AgentCore for hosted runtime. In June 2026 the core repo was consolidated into the strands-agents/harness-sdk monorepo (Python + TypeScript), repositioned around AWS's 'agent harness' framing; package names (strands-agents on PyPI) are unchanged.", + "tags": [ + "model-driven", + "aws", + "open-source" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 84, + "security": 78, + "privacy_compliance": 84, + "trust_transparency": 86, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "superagi", + "type": "agent", + "name": "SuperAGI", + "provider": "Community", + "description": "UNMAINTAINED: the SuperAGI repository is dormant — the last commit was January 2025 (an IDOR security fix) and the last release was v0.0.14 in January 2024; it is not recommended for new deployments. Formerly an open-source autonomous AI agent framework for running and managing multiple AI agents concurrently, with GUI-based management, a tool marketplace, and agent templates.", + "tags": [ + "autonomous", + "self-hosted", + "open-source", + "unmaintained" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 73, + "dimensions": { + "performance_reliability": 76, + "security": 70, + "privacy_compliance": 78, + "trust_transparency": 74, + "operational_excellence": 67 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "swarm", + "type": "agent", + "name": "OpenAI Swarm", + "provider": "OpenAI", + "description": "DEPRECATED: marked deprecated since 2025-03-11; the README redirects users to the OpenAI Agents SDK as the production successor (repo remains public and unmaintained, not GitHub-archived). Swarm was an experimental educational framework from OpenAI for multi-agent orchestration, demonstrating agent coordination and handoff patterns with simple Python primitives. Not maintained; do not use for new projects.", + "tags": [ + "openai", + "assistants", + "experimental", + "open-source", + "deprecated" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 74, + "dimensions": { + "performance_reliability": 73, + "security": 70, + "privacy_compliance": 75, + "trust_transparency": 82, + "operational_excellence": 69 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "warp", + "type": "agent", + "name": "Warp / Oz", + "provider": "Warp", + "description": "Warp's Agentic Development Environment (ADE): a Rust-based terminal reimagined for prompt-driven, multi-agent software development, paired with Oz, its cloud agent orchestration platform. Warp 2.0 launched the ADE in June 2025; the client was open-sourced under dual MIT/AGPLv3 licensing on 2026-04-28 with OpenAI as founding repository sponsor. Used by nearly a million developers, including Docker and over half the Fortune 500.", + "tags": [ + "agentic-development-environment", + "terminal", + "multi-agent", + "open-source", + "cloud-agents" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 81, + "security": 75, + "privacy_compliance": 70, + "trust_transparency": 83, + "operational_excellence": 78 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "zapier-ai", + "type": "agent", + "name": "Zapier AI Actions", + "provider": "Zapier", + "description": "No-code automation platform with AI actions for connecting AI models to 8,000+ apps. Enables building intelligent automations and AI-powered workflows without coding through visual interface and natural language. Now includes Zapier Agents (agents.zapier.com) for autonomous AI agents, Chatbots, and Zapier MCP for connecting external AI assistants to its app ecosystem.", + "tags": [ + "automation", + "workflow" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 86, + "dimensions": { + "performance_reliability": 84, + "security": 87, + "privacy_compliance": 86, + "trust_transparency": 86, + "operational_excellence": 87 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-apify", + "type": "mcp", + "name": "Apify MCP Server", + "provider": "Apify", + "description": "Official Apify MCP server (renamed from actors-mcp-server) exposing thousands of Apify Store Actors - scrapers for social media, maps, e-commerce, and search - via the @apify/actors-mcp-server package (v0.11.x) or the hosted endpoint at mcp.apify.com (OAuth or Bearer API token, Streamable HTTP only; legacy SSE removed April 2026). The hosted server adds output-schema inference, agentic payments (x402 USDC on Base, Skyfire), and June 2026 MCP connectors for login-required apps.", + "tags": [ + "web-scraping", + "automation", + "actors", + "data-extraction", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 81, + "security": 71, + "privacy_compliance": 66, + "trust_transparency": 85, + "operational_excellence": 84 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-asana", + "type": "mcp", + "name": "MCP Asana Server", + "provider": "Asana (Official)", + "description": "Asana's official hosted MCP server, now V2 at https://mcp.asana.com/v2/mcp (GA 2026-02-04, Streamable HTTP, OAuth with pre-registered apps, workspace-scoped authorization); the v1/beta SSE server was shut down 2026-05-11. Exposes task, project, and status tools. Carries a notable security history: in June 2025 a flawed tenant-isolation check in the beta server exposed data of ~1,000 organizations to other tenants (Jun 5-17), prompting a two-week outage and a rearchitected V2.", + "tags": [ + "project-management", + "task-tracking", + "mcp", + "model-context-protocol", + "official", + "remote-server", + "security-incident-history" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 71, + "dimensions": { + "performance_reliability": 82, + "security": 61, + "privacy_compliance": 62, + "trust_transparency": 70, + "operational_excellence": 80 + }, + "strengths_count": 5, + "limitations_count": 7 + }, + { + "id": "mcp-server-atlassian", + "type": "mcp", + "name": "MCP Atlassian Server", + "provider": "Atlassian", + "description": "Official Atlassian (Rovo) MCP server, generally available since February 2026 as a hosted remote at https://mcp.atlassian.com/v1/mcp (OAuth 2.1 via /v1/mcp/authv2, or API-token auth for machine-to-machine use). Connects Jira, Confluence, Jira Service Management, Bitbucket, and Compass to AI models for project management, issue tracking, and knowledge base operations. The legacy /v1/sse endpoint is unsupported after 2026-06-30.", + "tags": [ + "project-management", + "jira", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 83, + "security": 74, + "privacy_compliance": 71, + "trust_transparency": 85, + "operational_excellence": 82 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-aws", + "type": "mcp", + "name": "MCP AWS Server", + "provider": "AWS", + "description": "AWS's official MCP offering for AWS cloud services. Two forms: the managed AWS MCP Server (GA 2026-05-06), a fully managed remote exposing 15,000+ AWS API operations plus documentation search and a sandboxed run_script tool, accessed through the open-source MCP Proxy for AWS which bridges IAM SigV4 credentials to MCP's OAuth model; and the awslabs/mcp suite of open-source, service-scoped servers (DynamoDB, CDK, pricing, etc.). Powerful but requires strict IAM controls.", + "tags": [ + "aws", + "cloud", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 85, + "security": 67, + "privacy_compliance": 66, + "trust_transparency": 85, + "operational_excellence": 79 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-azure", + "type": "mcp", + "name": "MCP Azure Server", + "provider": "Microsoft", + "description": "Official Microsoft MCP server for Azure cloud services, generally available as Azure MCP Server 2.0 (2026-04-10) with tools spanning 44+ Azure service areas including compute, storage, databases, AI/ML, monitoring, and governance. Development now lives in the microsoft/mcp monorepo (the former Azure/azure-mcp repo is archived). Installable via npm (@azure/mcp), MCPB bundles for Claude Desktop, VS Code, NuGet, PyPI, and Docker, with a self-hosted remote HTTP mode.", + "tags": [ + "azure", + "microsoft", + "cloud", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 84, + "security": 72, + "privacy_compliance": 70, + "trust_transparency": 86, + "operational_excellence": 81 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "mcp-server-box", + "type": "mcp", + "name": "MCP Box Server", + "provider": "Box (Official)", + "description": "Box's official MCP server, a hosted remote at https://mcp.box.com using OAuth 2.0 with admin-managed enablement (Admin Console > Integrations) - the only supported path, as the self-hosted community Python server is deprecated. Tools cover user info, file/folder operations (read, list, search), and Box AI (Q&A across files, metadata extraction) over enterprise content. Mostly read plus AI extraction; permission-scoped to the authorizing user.", + "tags": [ + "file-storage", + "enterprise-content", + "document-ai", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 80, + "security": 75, + "privacy_compliance": 68, + "trust_transparency": 78, + "operational_excellence": 77 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-brave-search", + "type": "mcp", + "name": "MCP Brave Search Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED: Former Anthropic reference MCP server for the Brave Search API, archived 2025-05-29 to the servers-archived repository and no longer maintained (no security guarantees). Brave now ships its own official Brave Search MCP server (@brave/brave-search-mcp-server on npm, v2.0.85 as of June 2026), which is the recommended replacement. The underlying Brave Search API remains active and privacy-focused.", + "tags": [ + "search", + "web", + "mcp", + "model-context-protocol", + "archived", + "unmaintained" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 80, + "security": 84, + "privacy_compliance": 89, + "trust_transparency": 77, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-browserbase", + "type": "mcp", + "name": "Browserbase MCP Server", + "provider": "Browserbase", + "description": "Official Browserbase MCP server for cloud browser automation, powered by Stagehand v3: navigate, act, observe, extract (including iframes and shadow DOM), screenshots, and multi-session management. Hosted at https://mcp.browserbase.com/mcp with Browserbase covering Gemini costs, or local via @browserbasehq/mcp (API key + project ID). Stagehand v3 adds 20-40% faster automation via caching. Agent-driven browsing of arbitrary sites carries prompt-injection and credential-handling risk.", + "tags": [ + "browser-automation", + "cloud-browsers", + "stagehand", + "web-scraping", + "browserbase", + "mcp", + "model-context-protocol", + "official" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 75, + "dimensions": { + "performance_reliability": 85, + "security": 65, + "privacy_compliance": 57, + "trust_transparency": 85, + "operational_excellence": 85 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-calendar", + "type": "mcp", + "name": "MCP Google Calendar Server", + "provider": "Community", + "description": "Community-maintained MCP server for Google Calendar integration. Enables AI models to create, read, update, and delete calendar events, manage attendees, set reminders, and query availability. NOTE: Google now offers an official hosted Calendar MCP server (calendarmcp.googleapis.com) via the Workspace Developer Preview Program, announced May 2026; it supersedes community Calendar MCP servers for most new deployments and is the recommended option where available.", + "tags": [ + "calendar", + "scheduling", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 78, + "dimensions": { + "performance_reliability": 85, + "security": 73, + "privacy_compliance": 68, + "trust_transparency": 80, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-canva", + "type": "mcp", + "name": "Canva MCP Server (AI Connector)", + "provider": "Canva", + "description": "Canva's official MCP server, branded the AI Connector, hosted at https://mcp.canva.com/mcp over streamable HTTP with per-user OAuth (CIMD supported; custom clients register redirect URIs via an allowlist). Around 32 tools cover design generation and editing, library search, asset upload, folders, comments, exports (PDF/PNG/JPG/PPTX/MP4), resize, and brand-template autofill. Feature-gated by plan: resize needs Pro+, autofill and brand kits need Enterprise. Closed source.", + "tags": [ + "design", + "content-creation", + "canva", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 74, + "dimensions": { + "performance_reliability": 76, + "security": 73, + "privacy_compliance": 72, + "trust_transparency": 69, + "operational_excellence": 79 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-chrome-devtools", + "type": "mcp", + "name": "Chrome DevTools MCP", + "provider": "Google (Chrome DevTools team)", + "description": "Google's official MCP server that gives AI agents full Chrome control via the Chrome DevTools Protocol. Around 26 tools span input automation, navigation, performance tracing and insights, network inspection, console debugging, and screenshots — making it the reference server for AI-assisted web debugging and performance work.", + "tags": [ + "browser-automation", + "chrome", + "devtools", + "performance", + "debugging", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 87, + "security": 52, + "privacy_compliance": 65, + "trust_transparency": 92, + "operational_excellence": 90 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-clickhouse", + "type": "mcp", + "name": "MCP ClickHouse Server", + "provider": "ClickHouse (Official)", + "description": "Official ClickHouse MCP family: open-source mcp-clickhouse (PyPI, v0.4.0) plus ClickHouse Cloud Remote MCP at https://mcp.clickhouse.cloud/mcp with OAuth 2.0 and a Managed ClickStack MCP endpoint for observability. The local server is read-only by default; writes require CLICKHOUSE_ALLOW_WRITE_ACCESS=true, and destructive operations (DROP/TRUNCATE) need a second opt-in flag, CLICKHOUSE_ALLOW_DROP=true. Cloud remote tools are strictly read-only (readOnlyHint).", + "tags": [ + "database", + "analytics", + "olap", + "observability", + "sql", + "mcp", + "model-context-protocol", + "clickhouse" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 83, + "security": 78, + "privacy_compliance": 70, + "trust_transparency": 82, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-cloudflare", + "type": "mcp", + "name": "MCP Cloudflare Server", + "provider": "Cloudflare", + "description": "Official Cloudflare catalog of managed remote MCP servers, connected over OAuth. The primary Cloudflare API server at https://mcp.cloudflare.com/mcp exposes 2,500+ API endpoints (DNS, Workers, R2, Zero Trust, and more) through two Code Mode tools, search() and execute(), in roughly 1,000 tokens. Sixteen additional product-specific servers cover documentation, Workers bindings and builds, observability, Radar, browser rendering, AI Gateway, audit logs, and more.", + "tags": [ + "cloudflare", + "cdn", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 86, + "security": 73, + "privacy_compliance": 72, + "trust_transparency": 86, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-context7", + "type": "mcp", + "name": "Context7 MCP", + "provider": "Upstash", + "description": "Upstash's documentation-retrieval MCP server. Two tools (resolve-library-id, get-library-docs) inject up-to-date, version-specific library documentation into the agent's context to prevent hallucinated APIs. The most-starred MCP server repo (58.8k), but with a notable security history: the ContextCrush content-injection vulnerability (disclosed Feb 2026, patched within days).", + "tags": [ + "documentation", + "code-context", + "retrieval", + "upstash", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 86, + "security": 59, + "privacy_compliance": 75, + "trust_transparency": 86, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-databricks", + "type": "mcp", + "name": "MCP Databricks Managed Servers", + "provider": "Databricks (Official)", + "description": "Databricks managed MCP servers: workspace-hosted endpoints under https:///api/2.0/mcp/ for Genie natural-language data queries, AI Search (formerly vector-search; the legacy /api/2.0/mcp/vector-search prefix still works), Databricks SQL execution, and Unity Catalog functions as tools. OAuth with per-server scopes; Unity Catalog enforces permissions on every tool call and traffic is monitorable via the AI Gateway. Zero infrastructure — Databricks hosts and manages auth.", + "tags": [ + "database", + "lakehouse", + "data-warehouse", + "vector-search", + "nl-to-sql", + "mcp", + "model-context-protocol", + "databricks", + "managed" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 81, + "dimensions": { + "performance_reliability": 82, + "security": 82, + "privacy_compliance": 81, + "trust_transparency": 75, + "operational_excellence": 84 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-datadog", + "type": "mcp", + "name": "MCP Datadog Server", + "provider": "Datadog", + "description": "Official Datadog MCP server for observability integration, generally available since March 2026 as a hosted streamable-HTTP remote (US1: https://mcp.datadoghq.com/api/unstable/mcp-server/mcp, with per-site regional endpoints) plus a local binary. Authenticates via OAuth 2.0 (recommended), access-token bearer header, or DD_API_KEY/DD_APPLICATION_KEY headers. Queries metrics, logs, traces, dashboards, and alerts; apm, code-exec, and remote-actions toolsets remain in preview.", + "tags": [ + "monitoring", + "observability", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 88, + "security": 73, + "privacy_compliance": 68, + "trust_transparency": 86, + "operational_excellence": 87 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-docker", + "type": "mcp", + "name": "MCP Docker Server", + "provider": "Community", + "description": "Community-maintained MCP server for Docker container management; the leading implementation is ckreiling/mcp-server-docker (Python, GPL-3.0, ~728 stars), which manages containers via the Docker SDK and deliberately blocks sensitive options like --privileged and --cap-add. Covers containers, images, volumes, and networks. Note: Docker Inc. has NOT shipped an official engine-management MCP server; its official MCP Catalog, Toolkit, and Gateway distribute containerized MCP servers.", + "tags": [ + "containers", + "devops", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 85, + "security": 65, + "privacy_compliance": 68, + "trust_transparency": 80, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-elasticsearch", + "type": "mcp", + "name": "MCP Elasticsearch Server", + "provider": "Elastic", + "description": "Official Elastic MCP server (elastic/mcp-server-elasticsearch) exposing read-oriented tools - list_indices, get_mappings, search, esql, get_shards - for full-text search, ES|QL, and analytics. IMPORTANT: the standalone server is deprecated (v0.4.6, 2025-10-24, critical security updates only); Elastic directs users to the Agent Builder MCP endpoint ({KIBANA_URL}/api/agent_builder/mcp), GA in Elastic 9.2+ and Elasticsearch Serverless, with API key authentication.", + "tags": [ + "search", + "analytics", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 84, + "security": 75, + "privacy_compliance": 67, + "trust_transparency": 79, + "operational_excellence": 76 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-everything", + "type": "mcp", + "name": "MCP Everything Server", + "provider": "Anthropic", + "description": "Official MCP reference server demonstrating all protocol features (tools, resources, prompts, sampling) for testing and development. One of seven reference servers still maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest stable spec 2025-11-25 (2026-07-28 revision at release-candidate stage). Not for production use but essential for MCP protocol testing.", + "tags": [ + "meta", + "all-in-one", + "mcp", + "model-context-protocol", + "official", + "reference-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 80, + "security": 65, + "privacy_compliance": 75, + "trust_transparency": 94, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-exa", + "type": "mcp", + "name": "MCP Exa Server", + "provider": "Exa Labs", + "description": "Exa's official MCP server (exa-labs/exa-mcp-server v3.2.1, MIT) exposing neural/semantic web search, code search, crawling/content fetch, company and people research, and deep research tools. Hosted at https://mcp.exa.ai/mcp with an unauthenticated free tier (3 QPS, 150 calls/day) or an Exa API key for full access; also runs locally via npx exa-mcp-server. Read-only, agent-optimized results with source URLs; Instant Search (Feb 2026) delivers sub-150ms latency.", + "tags": [ + "search", + "web", + "semantic-search", + "research", + "exa", + "mcp", + "model-context-protocol", + "official" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 86, + "security": 76, + "privacy_compliance": 74, + "trust_transparency": 86, + "operational_excellence": 88 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "mcp-server-fetch", + "type": "mcp", + "name": "MCP Fetch Server", + "provider": "Anthropic", + "description": "Official MCP reference server for fetching web content and converting HTML to markdown. Enables AI models to retrieve and process web pages, documentation, and online resources. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest stable spec 2025-11-25; 2026-07-28 spec revision at release-candidate stage as of 2026-07).", + "tags": [ + "http", + "api", + "mcp", + "model-context-protocol", + "official", + "reference-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 84, + "security": 72, + "privacy_compliance": 68, + "trust_transparency": 89, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-figma", + "type": "mcp", + "name": "Figma MCP Server", + "provider": "Figma", + "description": "Figma's official Dev Mode MCP server connecting AI coding tools to design files. Provides design-context extraction (code, variables, components), screenshots, metadata, Code Connect mapping, FigJam reading, and design generation onto the canvas (canvas writes expanded March 2026). Available as a hosted remote server (OAuth, all plans) or via the Figma desktop app (Dev/Full seat on paid plans). Still in beta as of July 2026; free during the beta with per-plan tool-call limits.", + "tags": [ + "design", + "design-to-code", + "figma", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 75, + "dimensions": { + "performance_reliability": 78, + "security": 74, + "privacy_compliance": 72, + "trust_transparency": 71, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-filesystem", + "type": "mcp", + "name": "MCP Filesystem Server", + "provider": "Anthropic", + "description": "Official MCP reference server providing AI models with controlled access to the local filesystem (file reading, writing, and directory operations). One of seven reference servers still maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest stable spec 2025-11-25 (2026-07-28 revision at release-candidate stage). Critical for file-based workflows but requires careful security configuration.", + "tags": [ + "file-system", + "local", + "mcp", + "model-context-protocol", + "official", + "reference-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 81, + "dimensions": { + "performance_reliability": 91, + "security": 68, + "privacy_compliance": 72, + "trust_transparency": 85, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-firecrawl", + "type": "mcp", + "name": "Firecrawl MCP Server", + "provider": "Firecrawl", + "description": "Official Firecrawl MCP server (firecrawl-mcp v3.x) providing web scraping, crawling, site mapping, web search, structured extraction, document parsing, a research agent, interactive browser sessions, and page-change monitors. Runs as an npm package over stdio or as the hosted server at https://mcp.firecrawl.dev/{API_KEY}/v2/mcp; a rate-limited keyless free tier (https://mcp.firecrawl.dev/v2/mcp) covers scrape, search, and interact, and hosted transport accepts OAuth bearer tokens.", + "tags": [ + "web-scraping", + "crawling", + "search", + "extraction", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 83, + "security": 67, + "privacy_compliance": 69, + "trust_transparency": 88, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-git", + "type": "mcp", + "name": "MCP Git Server", + "provider": "Anthropic", + "description": "Official MCP reference server for local Git operations (commits, branches, version control). One of the seven reference servers still maintained after the 2025-05-29 archival; governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09 (stable spec 2025-11-25; 2026-07-28 revision at RC stage). CVE-2025-68143/68144/68145 (path traversal, argument injection; published 2026-01-20) are fixed in 2025.9.25 and 2025.12.18; current release 2026.6.16 includes all fixes.", + "tags": [ + "git", + "version-control", + "mcp", + "model-context-protocol", + "official", + "reference-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 87, + "security": 70, + "privacy_compliance": 66, + "trust_transparency": 90, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-github", + "type": "mcp", + "name": "MCP GitHub Server", + "provider": "GitHub (formerly Anthropic)", + "description": "GitHub's OFFICIAL MCP server, successor to the archived Anthropic reference server. Open source (Go, MIT); binary/Docker or hosted remote at https://api.githubcopilot.com/mcp/ (GA 2025-09-04, OAuth 2.1+PKCE). Since v1.5.0 (2026-06) the local stdio server has built-in OAuth (no PAT needed); releases track the 2026-01-26 MCP spec. 50+ tools in configurable toolsets with read-only mode. Prompt-injection exfiltration risk (Invariant Labs, May 2025) requires least-privilege tokens.", + "tags": [ + "git", + "version-control", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 86, + "security": 76, + "privacy_compliance": 71, + "trust_transparency": 90, + "operational_excellence": 87 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "mcp-server-gitlab", + "type": "mcp", + "name": "MCP GitLab Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED: Former Anthropic reference MCP server for GitLab (merge requests, CI/CD pipelines, issues), archived 2025-05-29 to servers-archived and no longer maintained; no security guarantees for archived servers. GitLab now ships an OFFICIAL built-in MCP server (experiment in GitLab 18.3, beta since 18.6) at https:///api/v4/mcp with OAuth 2.0 Dynamic Client Registration; it requires GitLab Duo and a Premium/Ultimate tier and should be preferred.", + "tags": [ + "git", + "gitlab", + "devops", + "ci-cd", + "mcp", + "model-context-protocol", + "archived", + "unmaintained" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 81, + "dimensions": { + "performance_reliability": 80, + "security": 83, + "privacy_compliance": 76, + "trust_transparency": 87, + "operational_excellence": 80 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-gmail", + "type": "mcp", + "name": "MCP Gmail Server", + "provider": "Community", + "description": "Community-maintained MCP server for Gmail email operations. Enables AI models to read, send, search, label, and manage emails through the Gmail API. Includes support for attachments, threading, and advanced search. NOTE: Google now offers an official hosted Gmail MCP server (gmailmcp.googleapis.com) via the Workspace Developer Preview Program, announced May 2026; it supersedes community Gmail MCP servers for most new deployments and is the recommended option where available.", + "tags": [ + "email", + "google", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 75, + "dimensions": { + "performance_reliability": 83, + "security": 70, + "privacy_compliance": 64, + "trust_transparency": 78, + "operational_excellence": 80 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-google-drive", + "type": "mcp", + "name": "MCP Google Drive Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED: Former Anthropic reference MCP server (gdrive) for accessing and searching Google Drive files, archived 2025-05-29 to servers-archived and no longer maintained; no security guarantees. Google now offers an official hosted Google Drive MCP server (drivemcp.googleapis.com) via the Workspace Developer Preview Program, announced May 2026, which is the recommended replacement. The archived server receives no fixes and is not recommended for new deployments.", + "tags": [ + "storage", + "google", + "mcp", + "model-context-protocol", + "archived", + "unmaintained" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 78, + "dimensions": { + "performance_reliability": 85, + "security": 78, + "privacy_compliance": 73, + "trust_transparency": 77, + "operational_excellence": 76 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-grafana", + "type": "mcp", + "name": "MCP Grafana Server", + "provider": "Grafana Labs", + "description": "Official Grafana Labs MCP server (grafana/mcp-grafana, Apache-2.0, Go) with 40+ tools: query metrics/logs/traces across Prometheus, Loki, and many other datasources, search and update dashboards, manage alert rules, and drive Incident, Sift, and OnCall. Supports stdio, SSE, and streamable HTTP with service-account token auth, category-level tool enablement, and a --disable-write read-only mode; also powers Grafana Assistant, with a hosted Grafana Cloud MCP endpoint in preview.", + "tags": [ + "monitoring", + "observability", + "grafana", + "prometheus", + "loki", + "mcp", + "model-context-protocol", + "official" + ], + "last_evaluated": "2026-07-09", + "release_year": 2026, + "overall_score": 82, + "dimensions": { + "performance_reliability": 87, + "security": 73, + "privacy_compliance": 72, + "trust_transparency": 90, + "operational_excellence": 88 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-hubspot", + "type": "mcp", + "name": "MCP HubSpot Server", + "provider": "HubSpot (Official)", + "description": "HubSpot's official hosted MCP server at https://mcp.hubspot.com, authenticated via OAuth 2.1 with PKCE and available to all HubSpot accounts and tiers. Launched as a read-only public beta in September 2025; GA on 2026-04-13 added write capabilities (CRM records and activities), activity history, marketing content objects, and organizational context - a material expansion of the risk surface. Actions inherit the connecting user's HubSpot permissions.", + "tags": [ + "crm", + "marketing", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 81, + "security": 73, + "privacy_compliance": 66, + "trust_transparency": 78, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-hugging-face", + "type": "mcp", + "name": "Hugging Face MCP Server", + "provider": "Hugging Face", + "description": "Hugging Face's official MCP server connecting AI assistants to the Hub. Ships 7 built-in tools (search for models, datasets, Spaces, and papers, plus documentation search) and can dynamically attach community Gradio Spaces as additional tools. Hosted at huggingface.co/mcp with per-user configuration, or runnable locally; open source under MIT.", + "tags": [ + "machine-learning", + "models", + "datasets", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 80, + "security": 71, + "privacy_compliance": 69, + "trust_transparency": 83, + "operational_excellence": 81 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-kubernetes", + "type": "mcp", + "name": "MCP Kubernetes Server", + "provider": "Community", + "description": "Community-maintained MCP servers for Kubernetes cluster management; there is still no single official upstream Kubernetes MCP server. Two implementations lead: containers/kubernetes-mcp-server (Go-native, talks directly to the Kubernetes API, supports OpenShift, Red Hat-backed, in the Red Hat Ecosystem Catalog) and Flux159/mcp-server-kubernetes (TypeScript, wraps kubectl/helm). Enables pod management, deployment orchestration, service configuration, and cluster resource monitoring.", + "tags": [ + "kubernetes", + "orchestration", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 84, + "security": 68, + "privacy_compliance": 69, + "trust_transparency": 79, + "operational_excellence": 81 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-linear", + "type": "mcp", + "name": "MCP Linear Server", + "provider": "Linear (official; community implementations also exist)", + "description": "Linear's OFFICIAL MCP server, a hosted remote at https://mcp.linear.app/mcp using OAuth 2.1 with dynamic client registration; Bearer-token/API-key auth is supported for non-interactive or read-only access. Exposes 25+ tools to create, read, update issues, manage projects, track cycles, and coordinate teams. Supersedes earlier community-maintained Linear MCP servers, which remain available as open-source alternatives.", + "tags": [ + "project-management", + "issue-tracking", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 86, + "security": 73, + "privacy_compliance": 69, + "trust_transparency": 81, + "operational_excellence": 84 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-memory", + "type": "mcp", + "name": "MCP Memory Server", + "provider": "Anthropic", + "description": "MCP reference server providing persistent memory and knowledge graph capabilities: long-term retention, entity relationship tracking, and contextual recall across conversations. One of the seven reference servers still maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09, latest stable spec 2025-11-25 (2026-07-28 revision at release-candidate stage). Critical for personalized AI but raises privacy concerns.", + "tags": [ + "memory", + "storage", + "mcp", + "model-context-protocol", + "official", + "reference-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 83, + "security": 72, + "privacy_compliance": 65, + "trust_transparency": 78, + "operational_excellence": 80 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-mongodb", + "type": "mcp", + "name": "MCP MongoDB Server", + "provider": "MongoDB (Official)", + "description": "Official MongoDB MCP server (mongodb-js/mongodb-mcp-server), generally available since v1.9.0 (2026-03-24). Enables AI models to query, insert, update, and delete documents, manage collections and indexes, run aggregations, and administer MongoDB Atlas clusters (50+ tools spanning database, Atlas, Atlas Local, and Assistant tools). Supports a read-only mode, elicitation-based user confirmation for destructive operations, and disables server-side JavaScript by default.", + "tags": [ + "database", + "nosql", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 83, + "security": 71, + "privacy_compliance": 66, + "trust_transparency": 80, + "operational_excellence": 82 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-neo4j", + "type": "mcp", + "name": "MCP Neo4j Server", + "provider": "Neo4j (Official)", + "description": "Official Neo4j MCP server (neo4j/mcp; neo4j-mcp-server on PyPI, v1.5.3 released 2026-06-11): schema introspection, read-cypher (read-only enforced via EXPLAIN query classification), write-cypher, and GDS procedure listing over stdio. Arbitrary Cypher means read+write access to the graph unless NEO4J_READ_ONLY=true is set. Distinct from the experimental Neo4j Labs servers (mcp-neo4j-cypher, mcp-neo4j-memory), which carry no product support or compatibility guarantees.", + "tags": [ + "database", + "graph", + "cypher", + "knowledge-graph", + "mcp", + "model-context-protocol", + "neo4j" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 74, + "dimensions": { + "performance_reliability": 80, + "security": 68, + "privacy_compliance": 68, + "trust_transparency": 78, + "operational_excellence": 78 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-neon", + "type": "mcp", + "name": "MCP Neon Server", + "provider": "Neon (Databricks)", + "description": "Official Neon MCP server for managing serverless Postgres via natural language: projects, branches, SQL execution, and branch-based safe migrations (20+ tools). Remote-only since Feb 2026 at https://mcp.neon.tech/mcp with OAuth (or API-key header); the local stdio CLI was removed and npm @neondatabase/mcp-server-neon (0.6.5) is deprecated. Full DDL/DML capability means Neon (Databricks-owned since May 2025) recommends development/testing use, not production databases.", + "tags": [ + "database", + "postgresql", + "serverless", + "neon", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 87, + "security": 77, + "privacy_compliance": 68, + "trust_transparency": 82, + "operational_excellence": 81 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-notion", + "type": "mcp", + "name": "MCP Notion Server", + "provider": "Notion (Official)", + "description": "Notion's official MCP integration. Primary offering is the hosted server at https://mcp.notion.com/mcp (Streamable HTTP and SSE, one-click OAuth), the only actively supported path. The open-source local @notionhq/notion-mcp-server (MIT, v2.4.1) is in maintenance mode: issues/PRs not actively monitored and Notion may sunset the repo. Enables creating, reading, updating pages and databases, managing blocks, and searching workspaces; recent releases track Notion API version 2026-03-11.", + "tags": [ + "productivity", + "notes", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 81, + "security": 74, + "privacy_compliance": 67, + "trust_transparency": 81, + "operational_excellence": 81 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-paypal", + "type": "mcp", + "name": "PayPal MCP Server", + "provider": "PayPal", + "description": "PayPal's official MCP server (Agent Toolkit), hosted at https://mcp.paypal.com with SSE (/sse) and streamable HTTP (/http) transports plus a sandbox at mcp.sandbox.paypal.com; also installable locally as @paypal/mcp. OAuth login or client-credential access tokens gate access, with restricted tool visibility so the LLM only sees tools its token permits. Tools cover invoices, orders, captures, refunds, disputes, subscriptions, catalog, shipment tracking, and transaction reporting.", + "tags": [ + "payments", + "invoicing", + "fintech", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 84, + "security": 78, + "privacy_compliance": 75, + "trust_transparency": 80, + "operational_excellence": 80 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-perplexity", + "type": "mcp", + "name": "MCP Perplexity Server", + "provider": "Perplexity AI", + "description": "Perplexity's official MCP server (perplexityai/modelcontextprotocol, npm @perplexity-ai/mcp-server) for the Perplexity API Platform. Exposes four tools — perplexity_search (Search API), perplexity_ask (sonar-pro), perplexity_research (sonar-deep-research), and perplexity_reason (sonar-reasoning-pro) — providing multi-source research with automatic citations and recency filtering for comprehensive information retrieval.", + "tags": [ + "search", + "research", + "citations", + "mcp", + "model-context-protocol", + "perplexity" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 88, + "security": 80, + "privacy_compliance": 78, + "trust_transparency": 88, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-playwright", + "type": "mcp", + "name": "Playwright MCP", + "provider": "Microsoft", + "description": "Microsoft's official MCP server for browser automation via Playwright. Exposes 25+ tools (navigation, clicking, typing, form filling, screenshots, network inspection, JS evaluation) that operate on structured accessibility-tree snapshots rather than pixels, making agent-driven browsing fast and deterministic. Supersedes the archived Puppeteer reference server.", + "tags": [ + "browser-automation", + "playwright", + "testing", + "web-scraping", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 80, + "dimensions": { + "performance_reliability": 88, + "security": 60, + "privacy_compliance": 69, + "trust_transparency": 92, + "operational_excellence": 90 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-postgres", + "type": "mcp", + "name": "MCP PostgreSQL Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED WITH UNPATCHED VULNERABILITY: Former Anthropic reference MCP server for PostgreSQL, archived 2025-05-29. A SQL injection flaw disclosed by Trend Micro (June 2025) remains unpatched because the repo is archived; the npm package (v0.6.2) is still published and downloaded. NOT RECOMMENDED for any use; a patched community fork (@zeddotdev/postgres-context-server, fixed in v0.1.4) and maintained alternatives (e.g., Microsoft's MCP server for Azure Database for PostgreSQL) exist.", + "tags": [ + "database", + "sql", + "mcp", + "model-context-protocol", + "archived", + "unmaintained", + "security-incidents" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 72, + "dimensions": { + "performance_reliability": 87, + "security": 51, + "privacy_compliance": 66, + "trust_transparency": 81, + "operational_excellence": 77 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-puppeteer", + "type": "mcp", + "name": "MCP Puppeteer Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED: Former Anthropic reference MCP server for Puppeteer browser automation, archived 2025-05-29 to the servers-archived repository and no longer maintained. The archived repo explicitly provides no security guarantees. For browser automation, Microsoft's Playwright MCP server (microsoft/playwright-mcp) is the recommended successor and remains actively maintained as of July 2026.", + "tags": [ + "browser", + "automation", + "mcp", + "model-context-protocol", + "archived", + "unmaintained" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 72, + "dimensions": { + "performance_reliability": 80, + "security": 63, + "privacy_compliance": 66, + "trust_transparency": 80, + "operational_excellence": 69 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-redis", + "type": "mcp", + "name": "MCP Redis Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED: Former Anthropic reference MCP server for Redis cache and data structure operations, archived 2025-05-29 to the servers-archived repository and no longer maintained (the archive README provides no security guarantees). Enabled AI models to interact with Redis for key-value operations, pub/sub messaging, lists, sets, sorted sets, and hashes. Redis now ships an official Redis MCP server (redis/mcp-redis, with redis/mcp-redis-cloud for Redis Cloud), which is the recommended replacement.", + "tags": [ + "cache", + "database", + "mcp", + "model-context-protocol", + "archived", + "unmaintained" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 88, + "security": 70, + "privacy_compliance": 68, + "trust_transparency": 77, + "operational_excellence": 80 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-s3", + "type": "mcp", + "name": "MCP S3 Server", + "provider": "Community", + "description": "MCP servers for AWS S3 storage operations, enabling AI models to upload, download, list, and manage objects in S3 buckets. The landscape has shifted to official AWS options: the managed AWS MCP Server (GA 2026-05) covers all S3 APIs via its call_aws tool, and awslabs ships a dedicated S3 Tables MCP Server (read-only by default, write enabled only with --allow-write). Community S3 servers (e.g. aws-samples/sample-mcp-server-s3) remain available but are no longer the recommended path.", + "tags": [ + "storage", + "aws", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 78, + "dimensions": { + "performance_reliability": 85, + "security": 73, + "privacy_compliance": 69, + "trust_transparency": 81, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-salesforce", + "type": "mcp", + "name": "MCP Salesforce Server", + "provider": "Salesforce (Official)", + "description": "Salesforce's official MCP offering. Primary path is Hosted MCP Servers (GA 2026-04-29, included for Enterprise Edition+ and Developer Edition orgs): Salesforce-managed endpoints exposing org data, flows, Apex actions, and Named Query APIs via OAuth with PKCE, enforcing CRUD/FLS/sharing as the authenticated user. Variants: local DX MCP server (@salesforce/mcp, v0.30.x, Apache-2.0) for dev workflows, and the Data 360 MCP server in developer preview (May 2026).", + "tags": [ + "crm", + "enterprise", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 84, + "security": 81, + "privacy_compliance": 68, + "trust_transparency": 82, + "operational_excellence": 81 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-sentry", + "type": "mcp", + "name": "MCP Sentry Server", + "provider": "Sentry", + "description": "Official Sentry MCP server (getsentry/sentry-mcp) for error tracking and monitoring integration. Sentry hosts a managed remote at https://mcp.sentry.dev/mcp with OAuth authentication (nothing to install); a stdio mode with an auth token supports self-hosted Sentry. Enables AI models to query errors, analyze stack traces, manage issues, track releases, and access performance data. Also distributed as a Claude Code plugin.", + "tags": [ + "error-tracking", + "monitoring", + "mcp", + "model-context-protocol", + "official", + "remote-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 87, + "security": 72, + "privacy_compliance": 66, + "trust_transparency": 84, + "operational_excellence": 85 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "mcp-server-sequential-thinking", + "type": "mcp", + "name": "MCP Sequential Thinking Server", + "provider": "Anthropic", + "description": "Official MCP reference server enabling dynamic, extended reasoning and problem-solving sequences: structured thinking processes, decomposition of complex problems, and context across multi-step reasoning chains. One of the seven reference servers still maintained after the 2025-05-29 archival; MCP governance moved to the Agentic AI Foundation (Linux Foundation) on 2025-12-09 (latest stable spec 2025-11-25; 2026-07-28 revision at release-candidate stage).", + "tags": [ + "reasoning", + "thinking", + "mcp", + "model-context-protocol", + "official", + "reference-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 83, + "dimensions": { + "performance_reliability": 82, + "security": 88, + "privacy_compliance": 80, + "trust_transparency": 85, + "operational_excellence": 78 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-serena", + "type": "mcp", + "name": "Serena MCP", + "provider": "Oraios AI", + "description": "Open-source semantic coding toolkit from Oraios AI that turns any MCP-capable agent into an IDE-grade coding assistant. Uses language servers (LSP) for symbol-level code navigation and editing — find_symbol, find_referencing_symbols, precise symbol edits, plus refactoring tools (rename/move/inline) — with project memory and shell execution across 40+ languages. An optional paid JetBrains-plugin backend adds interactive debugging. High-privilege local tooling: full filesystem and shell access.", + "tags": [ + "coding-agent", + "language-server", + "semantic-code", + "refactoring", + "local-tools", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 82, + "security": 49, + "privacy_compliance": 76, + "trust_transparency": 86, + "operational_excellence": 85 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "mcp-server-shadcn", + "type": "mcp", + "name": "shadcn MCP Server", + "provider": "shadcn", + "description": "MCP server built into the shadcn CLI (run via npx shadcn@latest mcp) that lets AI agents browse, search, and install UI components from any shadcn-compatible registry, including private registries, using @registry/name namespacing. Shipped with CLI 3.0 in August 2025; CLI v4 (March 2026) added shadcn/skills for coding agents, design-system presets, and --dry-run/--diff/--view flags to inspect registry changes before installation.", + "tags": [ + "ui-components", + "design-system", + "frontend", + "registry", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 82, + "dimensions": { + "performance_reliability": 85, + "security": 65, + "privacy_compliance": 83, + "trust_transparency": 88, + "operational_excellence": 88 + }, + "strengths_count": 6, + "limitations_count": 5 + }, + { + "id": "mcp-server-shopify", + "type": "mcp", + "name": "Shopify MCP Servers", + "provider": "Shopify", + "description": "Shopify's official MCP surface spans three layers: the Storefront MCP, live by default on every eligible store at {shop}.myshopify.com/api/mcp (public, no auth; catalog, cart, and policy tools), the local @shopify/dev-mcp stdio server (v1.14.x) for docs search and GraphQL schema work, and the open-source Shopify AI Toolkit (April 2026) adding authenticated admin operations via the Shopify CLI. Default-on endpoints across millions of stores form a huge aggregate agent surface.", + "tags": [ + "ecommerce", + "agentic-commerce", + "storefront", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 77, + "dimensions": { + "performance_reliability": 82, + "security": 66, + "privacy_compliance": 73, + "trust_transparency": 77, + "operational_excellence": 86 + }, + "strengths_count": 7, + "limitations_count": 7 + }, + { + "id": "mcp-server-slack", + "type": "mcp", + "name": "MCP Slack Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED: Former Anthropic reference MCP server for Slack workspace interaction, archived 2025-05-29 to the servers-archived repository. NO LONGER MAINTAINED and no security guarantees are provided for archived servers. Slack now ships an official Slack MCP server, generally available since 2026-02-17 (announced at Dreamforce October 2025) with expanded tools added 2026-05-13; it is the recommended replacement.", + "tags": [ + "communication", + "collaboration", + "mcp", + "model-context-protocol", + "archived", + "unmaintained" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 76, + "dimensions": { + "performance_reliability": 83, + "security": 75, + "privacy_compliance": 70, + "trust_transparency": 76, + "operational_excellence": 76 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-snowflake", + "type": "mcp", + "name": "MCP Snowflake Server", + "provider": "Snowflake (Official)", + "description": "Official Snowflake MCP server, GA since 2025-11-04 as a fully managed in-account endpoint (CREATE MCP SERVER object). Exposes Cortex Analyst NL-to-SQL over semantic views, Cortex Search, Cortex Agents, SQL execution with configurable read-only mode, and custom UDF/procedure tools. OAuth 2.0 or Programmatic Access Token auth with per-tool RBAC; masking and data policies apply. The self-hosted snowflake-labs-mcp (PyPI) was deprecated in May 2026 in favor of the managed server.", + "tags": [ + "database", + "data-warehouse", + "sql", + "nl-to-sql", + "mcp", + "model-context-protocol", + "snowflake", + "managed" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 81, + "dimensions": { + "performance_reliability": 84, + "security": 82, + "privacy_compliance": 80, + "trust_transparency": 76, + "operational_excellence": 83 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-sqlite", + "type": "mcp", + "name": "MCP SQLite Server", + "provider": "Anthropic (Archived)", + "description": "ARCHIVED WITH UNPATCHED VULNERABILITY: Former Anthropic reference MCP server for SQLite, archived 2025-05-29 to the servers-archived repository. The archived code contains the same class of unpatched SQL injection flaw disclosed by Trend Micro (June 2025) and will not receive fixes; the archive README provides no security guarantees. Not recommended for new deployments.", + "tags": [ + "database", + "sql", + "mcp", + "model-context-protocol", + "archived", + "unmaintained", + "security-incidents" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 89, + "security": 56, + "privacy_compliance": 73, + "trust_transparency": 91, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-stripe", + "type": "mcp", + "name": "Stripe MCP Server", + "provider": "Stripe", + "description": "Stripe's official MCP server (repo renamed from stripe/agent-toolkit to stripe/ai). The hosted server at mcp.stripe.com (OAuth or restricted-key bearer token) exposes generic API search/read/write tools plus resource search, docs search, an implementation planner, refunds, and account info, spanning payments, subscriptions, invoices, products, disputes, and balance. Treasury/payout tools are in preview by request. Also available as the @stripe/mcp stdio package (v0.3.3).", + "tags": [ + "payments", + "billing", + "fintech", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 86, + "security": 80, + "privacy_compliance": 77, + "trust_transparency": 89, + "operational_excellence": 86 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "mcp-server-supabase", + "type": "mcp", + "name": "MCP Supabase Server", + "provider": "Supabase", + "description": "Official Supabase MCP server: SQL execution, schema/migrations, type generation, Edge Function deploys, logs, storage, branching. Primarily a hosted remote at https://mcp.supabase.com/mcp with OAuth 2.1 dynamic client registration (PAT only for CI/CD), plus stdio and self-hosted modes. After 2025 prompt-injection/data-leak research, Supabase added mitigations: read-only mode, project_ref scoping, feature-group tool restrictions, and wrapping SQL results to deter embedded commands.", + "tags": [ + "database", + "postgresql", + "backend", + "mcp", + "model-context-protocol", + "supabase" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 84, + "dimensions": { + "performance_reliability": 85, + "security": 82, + "privacy_compliance": 80, + "trust_transparency": 88, + "operational_excellence": 86 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-tavily", + "type": "mcp", + "name": "MCP Tavily Server", + "provider": "Tavily", + "description": "Tavily's official MCP server (tavily-ai/tavily-mcp, MIT) enabling AI models to perform real-time web search, content extraction, site mapping, and web crawling. Available as a hosted remote endpoint at https://mcp.tavily.com/mcp/ (API key or OAuth) or locally via npx tavily-mcp. Designed specifically for AI agents with optimized search results and source verification capabilities.", + "tags": [ + "search", + "web", + "research", + "mcp", + "model-context-protocol", + "tavily" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 81, + "dimensions": { + "performance_reliability": 86, + "security": 78, + "privacy_compliance": 75, + "trust_transparency": 82, + "operational_excellence": 85 + }, + "strengths_count": 6, + "limitations_count": 6 + }, + { + "id": "mcp-server-time", + "type": "mcp", + "name": "MCP Time Server", + "provider": "Anthropic", + "description": "Official MCP reference server for time and timezone operations. Provides AI models with current time information, timezone conversions, date calculations, and scheduling assistance. One of the seven reference servers that remain actively maintained after the 2025-05-29 archival of non-core servers; MCP governance moved to the Agentic AI Foundation under the Linux Foundation on 2025-12-09 (latest stable spec 2025-11-25; 2026-07-28 spec revision at release-candidate stage as of 2026-07).", + "tags": [ + "time", + "utilities", + "mcp", + "model-context-protocol", + "official", + "reference-server" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 92, + "dimensions": { + "performance_reliability": 92, + "security": 95, + "privacy_compliance": 88, + "trust_transparency": 91, + "operational_excellence": 93 + }, + "strengths_count": 6, + "limitations_count": 7 + }, + { + "id": "mcp-server-vercel", + "type": "mcp", + "name": "Vercel MCP Server", + "provider": "Vercel", + "description": "Vercel's official hosted MCP server at mcp.vercel.com. Provides docs search plus tools to inspect teams, projects, deployments, build and runtime logs, and Agent Runs observability. No longer read-only: includes write-capable tools such as buy_domain (domain purchase), toolbar comment replies/edits/resolution, shareable-link creation for protected deployments, and CLI-guided deploys. Remote-only with OAuth 2.1, a client allowlist, and mandatory consent. Still in Beta on all plans.", + "tags": [ + "deployment", + "devops", + "hosting", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 83, + "security": 80, + "privacy_compliance": 76, + "trust_transparency": 74, + "operational_excellence": 82 + }, + "strengths_count": 7, + "limitations_count": 6 + }, + { + "id": "mcp-server-zapier", + "type": "mcp", + "name": "Zapier MCP Server", + "provider": "Zapier", + "description": "Zapier's hosted, proprietary MCP server giving AI agents access to user-selected actions from 9,000+ connected apps (40,000+ actions). Users create servers at mcp.zapier.com with per-app and per-action permissioning; listed clients connect via OAuth, unlisted clients use a rotatable connection token sent as an Authorization Bearer header (recommended) or embedded in the server URL. Transport is Streamable HTTP; each tool call consumes two Zapier tasks from the plan quota.", + "tags": [ + "automation", + "saas-integrations", + "workflow", + "hosted", + "mcp", + "model-context-protocol" + ], + "last_evaluated": "2026-07-09", + "release_year": null, + "overall_score": 79, + "dimensions": { + "performance_reliability": 85, + "security": 73, + "privacy_compliance": 75, + "trust_transparency": 73, + "operational_excellence": 87 + }, + "strengths_count": 7, + "limitations_count": 6 + } +]; diff --git a/lib/data.ts b/lib/data.ts index 96e07e2..7b622ec 100644 --- a/lib/data.ts +++ b/lib/data.ts @@ -1,443 +1,32 @@ /** * Data loading utilities for TrustVector + * + * SERVER-SIDE ONLY — do NOT import from 'use client' components. This module + * pulls in lib/data-index.ts, which bundles the FULL evaluation dataset + * (~4.4MB of JSON); importing it client-side ships all of it to the browser. + * Client list/compare views must use lib/client-data.ts (EntitySummary) + * instead. (We'd use `import 'server-only'` here, but that package is not a + * dependency of this project.) + * + * Entity JSON is bundled via lib/data-index.ts, which is auto-generated from + * the data/ directory by scripts/generate-data-index.ts (runs on predev/prebuild). */ import type { TrustVectorEntity } from '@/framework/schema/types'; import { calculateOverallScore } from '@/framework/schema/types'; +import { ALL_ENTITIES } from './data-index'; + +// Overall scores are constants for static data — compute once, not per sort comparison. +const OVERALL_SCORE_CACHE = new Map(); +function cachedOverallScore(entity: TrustVectorEntity): number { + let s = OVERALL_SCORE_CACHE.get(entity.id); + if (s === undefined) { + s = calculateOverallScore(entity); + OVERALL_SCORE_CACHE.set(entity.id, s); + } + return s; +} -// ======================================== -// MODELS (60 total) -// ======================================== - -// Anthropic Models (11) -import claudeFable5 from '@/data/models/claude-fable-5.json'; -import claudeOpus48 from '@/data/models/claude-opus-4-8.json'; -import claudeOpus47 from '@/data/models/claude-opus-4-7.json'; -import claudeOpus46 from '@/data/models/claude-opus-4-6.json'; -import claudeSonnet46 from '@/data/models/claude-sonnet-4-6.json'; -import claudeOpus45 from '@/data/models/claude-opus-4-5.json'; -import claudeSonnet45 from '@/data/models/claude-sonnet-4-5.json'; -import claudeSonnet4 from '@/data/models/claude-sonnet-4.json'; -import claudeOpus41 from '@/data/models/claude-opus-4-1.json'; -import claudeOpus4 from '@/data/models/claude-opus-4.json'; -import claudeHaiku45 from '@/data/models/claude-haiku-4-5.json'; - -// OpenAI Models (20) -import gpt55 from '@/data/models/gpt-5-5.json'; -import gpt54 from '@/data/models/gpt-5-4.json'; -import gpt53Codex from '@/data/models/gpt-5-3-codex.json'; -import gpt52 from '@/data/models/gpt-5-2.json'; -import gpt52Codex from '@/data/models/gpt-5-2-codex.json'; -import gpt51 from '@/data/models/gpt-5-1.json'; -import gpt5 from '@/data/models/gpt-5.json'; -import gpt41 from '@/data/models/gpt-4-1.json'; -import gpt41Mini from '@/data/models/gpt-4-1-mini.json'; -import gpt41Nano from '@/data/models/gpt-4-1-nano.json'; -import gpt4o from '@/data/models/gpt-4o.json'; -import gpt4oMini from '@/data/models/gpt-4o-mini.json'; -import openaiO1 from '@/data/models/openai-o1.json'; -import openaiO1Mini from '@/data/models/openai-o1-mini.json'; -import openaiO3 from '@/data/models/openai-o3.json'; -import openaiO3Mini from '@/data/models/openai-o3-mini.json'; -import openaiO4Mini from '@/data/models/openai-o4-mini.json'; -import gptOss120b from '@/data/models/gpt-oss-120b.json'; -import gptOss20b from '@/data/models/gpt-oss-20b.json'; - -// Google Models (8) -import gemini31Pro from '@/data/models/gemini-3-1-pro.json'; -import gemini35Flash from '@/data/models/gemini-3-5-flash.json'; -import gemma4 from '@/data/models/gemma-4.json'; -import gemini3Pro from '@/data/models/gemini-3-pro.json'; -import gemini3Flash from '@/data/models/gemini-3-flash.json'; -import gemini25Pro from '@/data/models/gemini-2-5-pro.json'; -import gemini20Flash from '@/data/models/gemini-2-0-flash.json'; -import gemma327b from '@/data/models/gemma-3-27b.json'; - -// Meta Models (5) -import llama4Maverick from '@/data/models/llama-4-maverick.json'; -import llama4Behemoth from '@/data/models/llama-4-behemoth.json'; -import llama4Scout from '@/data/models/llama-4-scout.json'; -import llama31405b from '@/data/models/llama-3-1-405b.json'; -import llama3370b from '@/data/models/llama-3-3-70b.json'; - -// xAI Models (3) -import grok43 from '@/data/models/grok-4-3.json'; -import grok41 from '@/data/models/grok-4-1.json'; -import grok3Beta from '@/data/models/grok-3-beta.json'; - -// DeepSeek Models (4) -import deepseekV4 from '@/data/models/deepseek-v4.json'; -import deepseekV32 from '@/data/models/deepseek-v3-2.json'; -import deepseekR1 from '@/data/models/deepseek-r1.json'; -import deepseekV30324 from '@/data/models/deepseek-v3-0324.json'; - -// Other Models (10) -import qwen35 from '@/data/models/qwen3-5.json'; -import kimiK26 from '@/data/models/kimi-k2-6.json'; -import glm5 from '@/data/models/glm-5.json'; -import minimaxM2 from '@/data/models/minimax-m2.json'; -import mistralLarge3 from '@/data/models/mistral-large-3.json'; -import commandAPlus from '@/data/models/command-a-plus.json'; -import nova2Lite from '@/data/models/nova-2-lite.json'; -import nemotronUltra253b from '@/data/models/nemotron-ultra-253b.json'; -import qwen25Vl32b from '@/data/models/qwen2-5-vl-32b.json'; -import novaPro from '@/data/models/nova-pro.json'; - -// ======================================== -// AGENTS (50 total) -// ======================================== - -// Coding & General-Purpose Agents (12, added 2026-06) -import claudeCode from '@/data/agents/claude-code.json'; -import claudeAgentSdk from '@/data/agents/claude-agent-sdk.json'; -import openaiAgentsSdk from '@/data/agents/openai-agents-sdk.json'; -import openaiCodex from '@/data/agents/openai-codex.json'; -import googleAdk from '@/data/agents/google-adk.json'; -import geminiCli from '@/data/agents/gemini-cli.json'; -import googleJules from '@/data/agents/google-jules.json'; -import githubCopilotCodingAgent from '@/data/agents/github-copilot-coding-agent.json'; -import microsoftAgentFramework from '@/data/agents/microsoft-agent-framework.json'; -import devin from '@/data/agents/devin.json'; -import cursorAgent from '@/data/agents/cursor-agent.json'; -import manus from '@/data/agents/manus.json'; - -// Open-Source Agent Frameworks (4, added 2026-06) -import smolagents from '@/data/agents/smolagents.json'; -import strandsAgents from '@/data/agents/strands-agents.json'; -import mastra from '@/data/agents/mastra.json'; -import dify from '@/data/agents/dify.json'; - -// Enterprise Agents (9) -import amazonLex from '@/data/agents/amazon-lex.json'; -import azureBotService from '@/data/agents/azure-bot-service.json'; -import googleDialogflow from '@/data/agents/google-dialogflow.json'; -import ibmWatsonAssistant from '@/data/agents/ibm-watson-assistant.json'; -import salesforceEinsteinBots from '@/data/agents/salesforce-einstein-bots.json'; -import gleanAi from '@/data/agents/glean-ai.json'; -import koreAi from '@/data/agents/kore-ai.json'; -import relevanceAi from '@/data/agents/relevance-ai.json'; -import sierraAi from '@/data/agents/sierra-ai.json'; - -// Cloud Provider Agents (3) -import openaiAssistants from '@/data/agents/openai-assistants-api.json'; -import bedrockAgents from '@/data/agents/amazon-bedrock-agents.json'; -import googleAgentBuilder from '@/data/agents/google-agent-builder.json'; - -// Open-Source Frameworks (8) -import rasa from '@/data/agents/rasa.json'; -import haystack from '@/data/agents/haystack.json'; -import langflow from '@/data/agents/langflow.json'; -import flowise from '@/data/agents/flowise.json'; -import superagi from '@/data/agents/superagi.json'; -import langgraphAgent from '@/data/agents/langgraph-agent.json'; -import llamaindexAgent from '@/data/agents/llamaindex-agent.json'; -import crewai from '@/data/agents/crewai.json'; - -// Microsoft Agents (2) -import autogen from '@/data/agents/autogen.json'; -import semanticKernel from '@/data/agents/semantic-kernel-agent.json'; - -// Workflow/Automation Agents (4) -import n8nAiAgent from '@/data/agents/n8n-ai-agent.json'; -import makeAi from '@/data/agents/make-ai.json'; -import zapierAi from '@/data/agents/zapier-ai.json'; -import activepieces from '@/data/agents/activepieces.json'; - -// Specialized Agents (8) -import agentgpt from '@/data/agents/agentgpt.json'; -import e2bAgents from '@/data/agents/e2b-agents.json'; -import pydanticAi from '@/data/agents/pydantic-ai.json'; -import swarm from '@/data/agents/swarm.json'; -import adala from '@/data/agents/adala.json'; -import memgpt from '@/data/agents/memgpt.json'; -import autogpt from '@/data/agents/autogpt.json'; -import babyagi from '@/data/agents/babyagi.json'; - -// ======================================== -// MCPs (46 total) -// ======================================== - -// Top Ecosystem MCPs (12, added 2026-06) -import mcpPlaywright from '@/data/mcps/mcp-server-playwright.json'; -import mcpChromeDevtools from '@/data/mcps/mcp-server-chrome-devtools.json'; -import mcpContext7 from '@/data/mcps/mcp-server-context7.json'; -import mcpSerena from '@/data/mcps/mcp-server-serena.json'; -import mcpFigma from '@/data/mcps/mcp-server-figma.json'; -import mcpStripe from '@/data/mcps/mcp-server-stripe.json'; -import mcpVercel from '@/data/mcps/mcp-server-vercel.json'; -import mcpHuggingFace from '@/data/mcps/mcp-server-hugging-face.json'; -import mcpFirecrawl from '@/data/mcps/mcp-server-firecrawl.json'; -import mcpShadcn from '@/data/mcps/mcp-server-shadcn.json'; -import mcpApify from '@/data/mcps/mcp-server-apify.json'; -import mcpZapier from '@/data/mcps/mcp-server-zapier.json'; - -// Official/Reference MCPs (5) -import mcpFetch from '@/data/mcps/mcp-server-fetch.json'; -import mcpGit from '@/data/mcps/mcp-server-git.json'; -import mcpSequentialThinking from '@/data/mcps/mcp-server-sequential-thinking.json'; -import mcpTime from '@/data/mcps/mcp-server-time.json'; -import mcpEverything from '@/data/mcps/mcp-server-everything.json'; - -// Search/AI MCPs (2) -import mcpPerplexity from '@/data/mcps/mcp-server-perplexity.json'; -import mcpTavily from '@/data/mcps/mcp-server-tavily.json'; - -// Version Control MCPs (2) -import mcpGitlab from '@/data/mcps/mcp-server-gitlab.json'; -import mcpSupabase from '@/data/mcps/mcp-server-supabase.json'; - -// Cloud Integration MCPs (5) -import mcpAws from '@/data/mcps/mcp-server-aws.json'; -import mcpAzure from '@/data/mcps/mcp-server-azure.json'; -import mcpCloudflare from '@/data/mcps/mcp-server-cloudflare.json'; -import mcpDocker from '@/data/mcps/mcp-server-docker.json'; -import mcpKubernetes from '@/data/mcps/mcp-server-kubernetes.json'; - -// Database MCPs (6) -import mcpPostgres from '@/data/mcps/mcp-server-postgres.json'; -import mcpMongodb from '@/data/mcps/mcp-server-mongodb.json'; -import mcpRedis from '@/data/mcps/mcp-server-redis.json'; -import mcpElasticsearch from '@/data/mcps/mcp-server-elasticsearch.json'; -import mcpSqlite from '@/data/mcps/mcp-server-sqlite.json'; -import mcpS3 from '@/data/mcps/mcp-server-s3.json'; - -// Productivity MCPs (8) -import mcpGmail from '@/data/mcps/mcp-server-gmail.json'; -import mcpCalendar from '@/data/mcps/mcp-server-calendar.json'; -import mcpNotion from '@/data/mcps/mcp-server-notion.json'; -import mcpLinear from '@/data/mcps/mcp-server-linear.json'; -import mcpAtlassian from '@/data/mcps/mcp-server-atlassian.json'; -import mcpSlack from '@/data/mcps/mcp-server-slack.json'; -import mcpGithub from '@/data/mcps/mcp-server-github.json'; -import mcpGoogleDrive from '@/data/mcps/mcp-server-google-drive.json'; - -// Developer Tools MCPs (4) -import mcpSentry from '@/data/mcps/mcp-server-sentry.json'; -import mcpDatadog from '@/data/mcps/mcp-server-datadog.json'; -import mcpPuppeteer from '@/data/mcps/mcp-server-puppeteer.json'; -import mcpBraveSearch from '@/data/mcps/mcp-server-brave-search.json'; - -// Utility MCPs (2) -import mcpFilesystem from '@/data/mcps/mcp-server-filesystem.json'; -import mcpMemory from '@/data/mcps/mcp-server-memory.json'; - -/** - * All entities in the system (156 total: 60 models + 50 agents + 46 MCPs) - */ -const ALL_ENTITIES: TrustVectorEntity[] = [ - // ======================================== - // MODELS (60) - // ======================================== - // Anthropic (11) - claudeFable5, - claudeOpus48, - claudeOpus47, - claudeOpus46, - claudeSonnet46, - claudeOpus45, - claudeSonnet45, - claudeSonnet4, - claudeOpus41, - claudeOpus4, - claudeHaiku45, - - // OpenAI (20) - gpt55, - gpt54, - gpt53Codex, - gpt52, - gpt52Codex, - gpt51, - gpt5, - gpt41, - gpt41Mini, - gpt41Nano, - gpt4o, - gpt4oMini, - openaiO1, - openaiO1Mini, - openaiO3, - openaiO3Mini, - openaiO4Mini, - gptOss120b, - gptOss20b, - - // Google (8) - gemini31Pro, - gemini35Flash, - gemma4, - gemini3Pro, - gemini3Flash, - gemini25Pro, - gemini20Flash, - gemma327b, - - // Meta (5) - llama4Maverick, - llama4Behemoth, - llama4Scout, - llama31405b, - llama3370b, - - // xAI (3) - grok43, - grok41, - grok3Beta, - - // DeepSeek (4) - deepseekV4, - deepseekV32, - deepseekR1, - deepseekV30324, - - // Other (10) - qwen35, - kimiK26, - glm5, - minimaxM2, - mistralLarge3, - commandAPlus, - nova2Lite, - nemotronUltra253b, - qwen25Vl32b, - novaPro, - - // ======================================== - // AGENTS (50) - // ======================================== - // Coding & General-Purpose Agents (12) - claudeCode, - claudeAgentSdk, - openaiAgentsSdk, - openaiCodex, - googleAdk, - geminiCli, - googleJules, - githubCopilotCodingAgent, - microsoftAgentFramework, - devin, - cursorAgent, - manus, - - // Open-Source Agent Frameworks — 2026 additions (4) - smolagents, - strandsAgents, - mastra, - dify, - - // Enterprise (9) - amazonLex, - azureBotService, - googleDialogflow, - ibmWatsonAssistant, - salesforceEinsteinBots, - gleanAi, - koreAi, - relevanceAi, - sierraAi, - - // Cloud Providers - openaiAssistants, - bedrockAgents, - googleAgentBuilder, - - // Open-Source - rasa, - haystack, - langflow, - flowise, - superagi, - langgraphAgent, - llamaindexAgent, - crewai, - - // Microsoft - autogen, - semanticKernel, - - // Workflow/Automation - n8nAiAgent, - makeAi, - zapierAi, - activepieces, - - // Specialized - agentgpt, - e2bAgents, - pydanticAi, - swarm, - adala, - memgpt, - autogpt, - babyagi, - - // ======================================== - // MCPs (46) - // ======================================== - // Top Ecosystem Servers — 2026 additions (12) - mcpPlaywright, - mcpChromeDevtools, - mcpContext7, - mcpSerena, - mcpFigma, - mcpStripe, - mcpVercel, - mcpHuggingFace, - mcpFirecrawl, - mcpShadcn, - mcpApify, - mcpZapier, - - // Official/Reference (5) - mcpFetch, - mcpGit, - mcpSequentialThinking, - mcpTime, - mcpEverything, - - // Search/AI (2) - mcpPerplexity, - mcpTavily, - - // Version Control/Database (2) - mcpGitlab, - mcpSupabase, - - // Cloud Integration - mcpAws, - mcpAzure, - mcpCloudflare, - mcpDocker, - mcpKubernetes, - - // Database - mcpPostgres, - mcpMongodb, - mcpRedis, - mcpElasticsearch, - mcpSqlite, - mcpS3, - - // Productivity - mcpGmail, - mcpCalendar, - mcpNotion, - mcpLinear, - mcpAtlassian, - mcpSlack, - mcpGithub, - mcpGoogleDrive, - - // Developer Tools - mcpSentry, - mcpDatadog, - mcpPuppeteer, - mcpBraveSearch, - - // Utility - mcpFilesystem, - mcpMemory, -] as TrustVectorEntity[]; /** * Get all entities @@ -512,7 +101,7 @@ export function filterEntities(filters: EntityFilters): TrustVectorEntity[] { if (filters.minOverallScore !== undefined) { results = results.filter((entity) => { - const overallScore = calculateOverallScore(entity); + const overallScore = cachedOverallScore(entity); return overallScore >= filters.minOverallScore!; }); } @@ -591,9 +180,9 @@ export function sortEntities( case 'name-desc': return sorted.sort((a, b) => b.name.localeCompare(a.name)); case 'overall-score-desc': - return sorted.sort((a, b) => calculateOverallScore(b) - calculateOverallScore(a)); + return sorted.sort((a, b) => cachedOverallScore(b) - cachedOverallScore(a)); case 'overall-score-asc': - return sorted.sort((a, b) => calculateOverallScore(a) - calculateOverallScore(b)); + return sorted.sort((a, b) => cachedOverallScore(a) - cachedOverallScore(b)); case 'date-desc': return sorted.sort((a, b) => b.last_evaluated.localeCompare(a.last_evaluated)); case 'date-asc': @@ -617,7 +206,7 @@ export function getStatistics() { agents: entities.filter((e) => e.type === 'agent').length, }, averageOverallScore: - entities.reduce((sum, entity) => sum + calculateOverallScore(entity), 0) / entities.length, + entities.reduce((sum, entity) => sum + cachedOverallScore(entity), 0) / entities.length, providers: getAllProviders().length, }; } diff --git a/lib/summary-types.ts b/lib/summary-types.ts new file mode 100644 index 0000000..c7a8d50 --- /dev/null +++ b/lib/summary-types.ts @@ -0,0 +1,34 @@ +/** + * Lightweight entity summary shape shared by the auto-generated + * lib/data-summaries.ts (see scripts/generate-data-index.ts) and its + * client-side consumers (lib/client-data.ts, list/compare views). + * + * Summaries carry only what list and compare views render (~5% of the full + * evaluation JSON), so client bundles never ship the full dataset. Detail + * pages are server components and read full entities via lib/data.ts. + */ + +export interface EntitySummary { + id: string; + type: 'model' | 'mcp' | 'agent'; + name: string; + provider: string; + description: string; + tags: string[]; + last_evaluated: string; + /** UTC year of metadata.release_date, or null when absent/unparseable. */ + release_year: number | null; + /** Rounded mean of the 5 trust_vector dimension overall_scores. */ + overall_score: number; + /** Per-dimension overall scores (null when a dimension is absent). */ + dimensions: { + performance_reliability: number | null; + security: number | null; + privacy_compliance: number | null; + trust_transparency: number | null; + operational_excellence: number | null; + }; + /** Counts only — full strengths/limitations text stays server-side. */ + strengths_count: number; + limitations_count: number; +} diff --git a/lib/utils.ts b/lib/utils.ts index 7192fba..d7a2d20 100644 --- a/lib/utils.ts +++ b/lib/utils.ts @@ -10,10 +10,13 @@ export function cn(...inputs: ClassValue[]) { */ export function formatDate(dateString: string): string { const date = new Date(dateString); + // Date-only strings parse as UTC midnight; format in UTC so + // '2026-07-09' never renders as July 8 in western timezones. return date.toLocaleDateString('en-US', { year: 'numeric', month: 'long', day: 'numeric', + timeZone: 'UTC', }); } diff --git a/package-lock.json b/package-lock.json index 19e1485..2855bc0 100644 --- a/package-lock.json +++ b/package-lock.json @@ -8567,6 +8567,21 @@ "funding": { "url": "https://github.com/sponsors/colinhacks" } + }, + "node_modules/@next/swc-win32-ia32-msvc": { + "version": "14.2.33", + "resolved": "https://registry.npmjs.org/@next/swc-win32-ia32-msvc/-/swc-win32-ia32-msvc-14.2.33.tgz", + "integrity": "sha512-pc9LpGNKhJ0dXQhZ5QMmYxtARwwmWLpeocFmVG5Z0DzWq5Uf0izcI8tLc+qOpqxO1PWqZ5A7J1blrUIKrIFc7Q==", + "cpu": [ + "ia32" + ], + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + } } } } diff --git a/package.json b/package.json index 4036d12..5c9cfd7 100644 --- a/package.json +++ b/package.json @@ -40,7 +40,12 @@ "lint": "next lint", "validate": "tsx scripts/validate-data.ts", "test": "vitest", - "type-check": "tsc --noEmit" + "type-check": "tsc --noEmit", + "generate:data-index": "tsx scripts/generate-data-index.ts", + "predev": "tsx scripts/generate-data-index.ts", + "prebuild": "tsx scripts/generate-data-index.ts", + "pretype-check": "tsx scripts/generate-data-index.ts", + "prevalidate": "tsx scripts/generate-data-index.ts" }, "dependencies": { "class-variance-authority": "^0.7.0", diff --git a/scripts/generate-data-index.ts b/scripts/generate-data-index.ts new file mode 100644 index 0000000..5b38455 --- /dev/null +++ b/scripts/generate-data-index.ts @@ -0,0 +1,190 @@ +#!/usr/bin/env tsx + +/** + * Generates lib/data-index.ts AND lib/data-summaries.ts from the JSON files + * in data/{models,agents,mcps}. + * + * - lib/data-index.ts statically imports every entity JSON. It is consumed + * ONLY by server-side code (detail pages, sitemap) via lib/data.ts, so the + * full ~4MB dataset never reaches the client bundle. + * - lib/data-summaries.ts embeds a small literal array of EntitySummary + * objects (~5% of the data) for the client-side list/compare views via + * lib/client-data.ts. + * + * Draft gate: files whose name starts with '_' or whose content has a + * top-level `"draft": true` are skipped from BOTH outputs. + * + * This script runs via `predev`/`prebuild`, so adding an evaluation file is + * all that's needed to publish it. + */ + +import { readdirSync, readFileSync, writeFileSync } from 'fs'; +import { join } from 'path'; + +import type { EntitySummary } from '../lib/summary-types'; + +const CATEGORIES = ['models', 'agents', 'mcps'] as const; + +const DIMENSIONS = [ + 'performance_reliability', + 'security', + 'privacy_compliance', + 'trust_transparency', + 'operational_excellence', +] as const; + +function identFor(category: string, file: string): string { + const base = file.replace(/\.json$/, ''); + const camel = base.replace(/[^a-zA-Z0-9]+(.)/g, (_, c: string) => c.toUpperCase()); + const safe = /^[a-zA-Z_]/.test(camel) ? camel : `_${camel}`; + return `${category}_${safe}`; +} + +/** Parse metadata.release_date to a UTC year, tolerating bad/missing input. */ +function releaseYearOf(entity: any): number | null { + const releaseDate = entity?.metadata?.release_date; + if (typeof releaseDate !== 'string' || releaseDate.length === 0) return null; + // Date-only strings ("YYYY-MM-DD") are parsed as UTC per the ECMAScript spec. + const parsed = new Date(releaseDate); + if (Number.isNaN(parsed.getTime())) return null; + return parsed.getUTCFullYear(); +} + +function dimensionScoreOf(entity: any, dimension: string): number | null { + const score = entity?.trust_vector?.[dimension]?.overall_score; + return typeof score === 'number' ? score : null; +} + +function summarize(entity: any): EntitySummary { + const dimensions = { + performance_reliability: dimensionScoreOf(entity, 'performance_reliability'), + security: dimensionScoreOf(entity, 'security'), + privacy_compliance: dimensionScoreOf(entity, 'privacy_compliance'), + trust_transparency: dimensionScoreOf(entity, 'trust_transparency'), + operational_excellence: dimensionScoreOf(entity, 'operational_excellence'), + }; + + // Rounded mean of the present dimension scores — mirrors + // framework/schema/types.ts calculateOverallScore (which averages + // Object.values(trust_vector)). + const present = DIMENSIONS.map((d) => dimensions[d]).filter( + (s): s is number => s !== null + ); + const overall_score = + present.length > 0 + ? Math.round(present.reduce((sum, s) => sum + s, 0) / present.length) + : 0; + + return { + id: entity.id, + type: entity.type, + name: entity.name, + provider: entity.provider, + description: entity.description, + tags: Array.isArray(entity.tags) ? entity.tags : [], + last_evaluated: entity.last_evaluated, + release_year: releaseYearOf(entity), + overall_score, + dimensions, + strengths_count: Array.isArray(entity.strengths) ? entity.strengths.length : 0, + limitations_count: Array.isArray(entity.limitations) ? entity.limitations.length : 0, + }; +} + +const imports: string[] = []; +const arrays: Record = {}; +const seen = new Map(); +const summaries: EntitySummary[] = []; +let total = 0; +let skippedDrafts = 0; + +for (const category of CATEGORIES) { + const dir = join(process.cwd(), 'data', category); + const files = readdirSync(dir) + .filter((f) => f.endsWith('.json')) + .sort(); + arrays[category] = []; + for (const file of files) { + // Draft gate #1: underscore-prefixed filenames are drafts. + if (file.startsWith('_')) { + console.log(`Skipping draft (underscore-prefixed): ${category}/${file}`); + skippedDrafts++; + continue; + } + + let entity: any; + try { + entity = JSON.parse(readFileSync(join(dir, file), 'utf8')); + } catch (err) { + console.error(`Failed to parse ${category}/${file}: ${err}`); + process.exit(1); + } + + // Draft gate #2: top-level `"draft": true` marks the file as a draft. + if (entity?.draft === true) { + console.log(`Skipping draft ("draft": true): ${category}/${file}`); + skippedDrafts++; + continue; + } + + const ident = identFor(category, file); + const prev = seen.get(ident); + if (prev) { + console.error( + `Identifier collision: ${category}/${file} and ${prev} both map to '${ident}'. Rename one file.` + ); + process.exit(1); + } + seen.set(ident, `${category}/${file}`); + imports.push(`import ${ident} from '@/data/${category}/${file}';`); + arrays[category].push(ident); + summaries.push(summarize(entity)); + total++; + } +} + +const indexOut = `/** + * AUTO-GENERATED by scripts/generate-data-index.ts — do not edit by hand. + * Regenerate with: npm run generate:data-index + * + * SERVER-SIDE ONLY: bundles the full evaluation dataset. Import via + * lib/data.ts from server components only — never from 'use client' code. + * Client list/compare views use lib/data-summaries.ts instead. + * + * ${total} entities: ${arrays.models.length} models, ${arrays.agents.length} agents, ${arrays.mcps.length} MCP servers. + */ + +import type { TrustVectorEntity } from '@/framework/schema/types'; + +${imports.join('\n')} + +export const ALL_ENTITIES: TrustVectorEntity[] = [ +${CATEGORIES.map( + (c) => ` // ${c} (${arrays[c].length})\n${arrays[c].map((i) => ` ${i},`).join('\n')}` +).join('\n')} +] as TrustVectorEntity[]; +`; + +const summariesOut = `/** + * AUTO-GENERATED by scripts/generate-data-index.ts — do not edit by hand. + * Regenerate with: npm run generate:data-index + * + * Lightweight summaries (~5% of the full dataset) embedded as a literal so + * client bundles carry only these small objects — no full evaluation JSON. + * Consumed by lib/client-data.ts. + * + * ${total} entities: ${arrays.models.length} models, ${arrays.agents.length} agents, ${arrays.mcps.length} MCP servers. + */ + +import type { EntitySummary } from '@/lib/summary-types'; + +export const ALL_SUMMARIES: EntitySummary[] = ${JSON.stringify(summaries, null, 2)}; +`; + +writeFileSync(join(process.cwd(), 'lib', 'data-index.ts'), indexOut); +writeFileSync(join(process.cwd(), 'lib', 'data-summaries.ts'), summariesOut); +console.log( + `Generated lib/data-index.ts + lib/data-summaries.ts: ${total} entities ` + + `(${arrays.models.length} models, ${arrays.agents.length} agents, ${arrays.mcps.length} MCPs)` + + (skippedDrafts > 0 ? `, ${skippedDrafts} draft(s) skipped` : '') +); From 789e3bced24d44937bdf12a3e870e8cdba012eef Mon Sep 17 00:00:00 2001 From: JBAhire Date: Thu, 9 Jul 2026 22:53:33 -0700 Subject: [PATCH 4/7] feat: migrate TrustVector to the Guard0 design system Emerald #10B981 primary on FAFAFA/true-black surfaces, Plus Jakarta Sans display + Inter body + JetBrains Mono (variable fonts vendored locally for offline builds), guard0 dog-mark logo lockup ('TrustVector by Guard0'), sentence-case hero matching guard0.ai, dot-grid, small radii, elevation + motion system with prefers-reduced-motion support. Scores render as a semantic traffic-light ramp (emerald/light-emerald/ amber/orange/red) from a single SCORE_THEMES source of truth; badges, bars, and segments all derive from it. Fixes: UTC date rendering (July 8 off-by-one), animation fill-mode pinning the card hover lift, duplicate box-shadow overriding the hover glow, search-icon overlap, stale sky-blue Strong leftovers, dead Loot Drop-era CSS. README badges, findings, and screenshots updated for the 196-entity registry. Co-Authored-By: Claude Fable 5 --- README.md | 51 ++-- app/agents/[id]/page.tsx | 100 +++---- app/compare/page.tsx | 84 +++--- app/contribute/page.tsx | 93 +++---- app/fonts/inter-var.woff2 | Bin 0 -> 48432 bytes app/fonts/jakarta-var.woff2 | Bin 0 -> 27272 bytes app/fonts/jetbrains-mono-var.woff2 | Bin 0 -> 40480 bytes app/globals.css | 430 ++++++++++++++--------------- app/layout.tsx | 63 +++-- app/mcps/[id]/page.tsx | 100 +++---- app/methodology/page.tsx | 62 ++--- app/models/[id]/page.tsx | 100 +++---- app/page.tsx | 99 +++---- components/entity-card.tsx | 58 ++-- components/export-pdf-button.tsx | 7 +- components/logo.tsx | 138 +++------ components/score-badge.tsx | 8 +- components/trust-vector-chart.tsx | 14 +- components/ui/badge.tsx | 2 +- framework/schema/types.ts | 33 ++- public/guard0-dog.png | Bin 0 -> 25686 bytes screenshots/details page.png | Bin 118579 -> 154751 bytes screenshots/header.png | Bin 84226 -> 82251 bytes tailwind.config.ts | 37 ++- 24 files changed, 683 insertions(+), 796 deletions(-) create mode 100644 app/fonts/inter-var.woff2 create mode 100644 app/fonts/jakarta-var.woff2 create mode 100644 app/fonts/jetbrains-mono-var.woff2 create mode 100644 public/guard0-dog.png diff --git a/README.md b/README.md index 0be4cf2..cfbb8d0 100644 --- a/README.md +++ b/README.md @@ -7,10 +7,10 @@ **Benchmarks tell you how smart an AI is. TrustVector tells you whether you can trust it in production.** [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) -[![Evaluations](https://img.shields.io/badge/Evaluations-156-blue.svg)](#-current-coverage) -[![Models](https://img.shields.io/badge/Models-60-8A2BE2.svg)](/data/models) -[![Agents](https://img.shields.io/badge/Agents-50-orange.svg)](/data/agents) -[![MCP Servers](https://img.shields.io/badge/MCP_Servers-46-green.svg)](/data/mcps) +[![Evaluations](https://img.shields.io/badge/Evaluations-196-blue.svg)](#-current-coverage) +[![Models](https://img.shields.io/badge/Models-68-8A2BE2.svg)](/data/models) +[![Agents](https://img.shields.io/badge/Agents-67-orange.svg)](/data/agents) +[![MCP Servers](https://img.shields.io/badge/MCP_Servers-61-green.svg)](/data/mcps) [![PRs Welcome](https://img.shields.io/badge/PRs-welcome-brightgreen.svg)](http://makeapullrequest.com) [![GitHub Stars](https://img.shields.io/github/stars/guard0-ai/TrustVector?style=social)](https://github.com/guard0-ai/TrustVector) @@ -22,15 +22,16 @@ --- -## 🚨 What our data found (June 2026) +## 🚨 What our data found (July 2026) This isn't a list of logos. Every entry is an evidence-linked evaluation across **security, privacy, performance, transparency, and operations** — and the findings are uncomfortable: -- ⚠️ **21 of 60 models** in the registry are **retired, deprecated, superseded, or were never released** — including models still hardcoded in thousands of production apps (Grok 3 now *silently redirects* to a different model; Gemini 2.0 Flash was shut down June 1). -- 🩸 **Archived MCP reference servers ship unpatched SQL injection.** The Postgres reference server was still pulling ~21k weekly downloads *after* being archived with a known SQLi — we score it 51/100 on security so you don't find out the hard way. -- 🕳️ **Popularity ≠ safety.** Context7 (57k★, the most-starred MCP server on GitHub) scores 86/100 on performance but **59/100 on security** after the "ContextCrush" registry-poisoning vulnerability. Playwright MCP: 88 performance, 60 security. -- 🔓 **The agent you let browse the web matters.** General-purpose autonomous agents score as low as **50/100 on privacy** in our registry; sandboxed, permission-gated coding agents score 20+ points higher. -- ⏳ **The OpenAI Assistants API sunsets August 26, 2026.** If you're on it, your migration window is measured in weeks. It's flagged. +- ⚠️ **35+ of 68 models** in the registry are **retired, deprecated, superseded, or were never released** — including models still hardcoded in thousands of production apps (Grok 3 now *silently redirects* to a different model — at that model's pricing; Gemini 3 Pro was retired just 4 months after launch; GPT-5 itself has a December 2026 shutdown date). +- 🚫 **Even the #1 model on the leaderboard can vanish.** Claude Fable 5 — the highest-scoring model in this registry — was suspended globally for 19 days in June 2026 under US export controls after a safeguard bypass was found. It's back, but our uptime and jailbreak scores now reflect it. +- 🩸 **Archived MCP reference servers still ship unpatched SQL injection.** The Postgres reference server was still pulling ~21k weekly downloads *after* being archived with a known SQLi — we score it 51/100 on security so you don't find out the hard way. Meanwhile Langflow, Flowise, and n8n all had critical RCEs actively exploited in 2026 (all patched — check your version). +- 🕳️ **Popularity ≠ safety.** Context7 (58k★, the most-starred MCP server on GitHub) scores 86/100 on performance but **59/100 on security** after the "ContextCrush" registry-poisoning vulnerability. Playwright MCP: 88 performance, 60 security. +- 🔓 **The agent you let into your life matters.** OpenClaw — the viral open-source assistant with 300K+ GitHub stars — scores **37/100 on security** (hundreds of CVEs, 135K+ exposed instances). ByteDance's free Trae IDE scores **24/100 on privacy** (telemetry that survives opt-out, 5-year retention). Sandboxed, permission-gated coding agents score 30-50 points higher on both. +- ⏳ **The OpenAI Assistants API sunsets August 26, 2026.** That's ~7 weeks out. If you're on it, your migration window is measured in weeks. It's flagged. **Every one of these claims links to a primary source with a date.** That's the whole point. @@ -42,19 +43,21 @@ Overall = mean of 5 dimension scores. Full criteria, evidence URLs, and confiden | Model | Overall | Perf | Security | Privacy | Transparency | Ops | |---|:---:|:---:|:---:|:---:|:---:|:---:| -| **Claude Fable 5** (Anthropic) | **92** | 98 | 92 | 93 | 88 | 91 | +| **Claude Fable 5** (Anthropic) | **92** | 96 | 90 | 93 | 88 | 91 | | **Claude Opus 4.8** (Anthropic) | **92** | 96 | 92 | 93 | 88 | 91 | +| **Claude Sonnet 5** (Anthropic) | **91** | 94 | 91 | 93 | 88 | 90 | | **GPT-5.5** (OpenAI) | **91** | 97 | 89 | 87 | 90 | 94 | | **Gemini 3.1 Pro** (Google) | **91** | 96 | 88 | 88 | 88 | 93 | +| **GPT-5.6 Sol** (OpenAI) | **89** | 94 | 88 | 87 | 87 | 90 | | **Mistral Large 3** (Mistral, open) | **85** | 88 | 83 | 87 | 80 | 86 | -| **Grok 4.3** (xAI) | **83** | 94 | 83 | 76 | 82 | 82 | +| **Grok 4.3** (SpaceXAI) | **83** | 94 | 82 | 76 | 81 | 82 | | **DeepSeek-V4** (open) | **83** | 92 | 83 | 78 | 80 | 83 | | **GLM-5** (Z.ai, open) | **82** | 92 | 80 | 75 | 81 | 83 | | **Kimi K2.6** (Moonshot, open) | **81** | 91 | 79 | 75 | 80 | 82 | Notice the spread: models within 5 points of each other on *capability* differ by **15+ points on privacy and security**. If you're choosing a model for healthcare, legal, or finance, the right-hand columns are the ones that get you fired. -And it's not just models — the same lens on **coding agents** (Claude Code 80, OpenAI Codex 82, Devin 71, Manus 64) and **MCP servers** (GitHub 82, Playwright 80, Context7 79, archived Postgres 72) exposes exactly where the trust gaps are. +And it's not just models — the same lens on **agents** (Claude Code 80, OpenAI Codex 82, Devin 71, Claude Cowork 73, ChatGPT Agent 69, OpenClaw 60) and **MCP servers** (GitHub 82, Snowflake 81, Playwright 80, Asana 71, archived Postgres 72) exposes exactly where the trust gaps are.
TrustVector detail page — per-criterion scores with evidence @@ -130,22 +133,26 @@ Predefined profiles: `balanced` · `security_first` · `performance_focused` · ## 📦 Current Coverage -**156 evaluations** across 3 categories (last refreshed June 2026 — yes, including the models that launched *this month*): +**196 evaluations** across 3 categories (last refreshed July 9, 2026 — every file re-verified against primary sources, same-day as the Grok 4.5 and GPT-5.6 launches):
-🧠 AI Models (60) — Claude Fable 5 → archived also-rans, all scored +🧠 AI Models (68) — Claude Fable 5 → archived also-rans, all scored -**Frontier:** Claude Fable 5, Opus 4.8/4.7/4.6/4.5, Sonnet 4.6/4.5, Haiku 4.5 · GPT-5.5, GPT-5.4, GPT-5.3-Codex, GPT-5.2, GPT-5.1, GPT-5, o-series · Gemini 3.1 Pro, Gemini 3.5 Flash, Gemini 3 · Grok 4.3, Grok 4.1 · Nova 2 Lite, Nova Pro +**Frontier:** Claude Fable 5, Sonnet 5, Opus 4.8/4.7/4.6/4.5, Sonnet 4.6/4.5, Haiku 4.5 · GPT-5.6 (Sol/Terra/Luna), GPT-5.5, GPT-5.4, GPT-5.3-Codex, GPT-5.2, GPT-5.1, GPT-5, o-series · Gemini 3.1 Pro, Gemini 3.5 Flash, Gemini 3 · Grok 4.5, Grok 4.3, Grok 4.1 · Nova 2 Lite, Nova Pro -**Open-weight:** DeepSeek V4 / V3.2 / R1 · Qwen3.5 · Kimi K2.6 · GLM-5 · MiniMax-M2 · Mistral Large 3 · Command A+ · Gemma 4 / 3 · gpt-oss-120b/20b · Llama 4 / 3.x · Nemotron +**Open-weight:** DeepSeek V4 / V3.2 / R1 · Qwen3.6 / 3.5 · Kimi K2.7-Code / K2.6 · GLM-5.2 / 5 · MiniMax M3 / M2 · Mistral Large 3 · Command A+ · Gemma 4 / 3 · gpt-oss-120b/20b · Llama 4 / 3.x · Nemotron 3 Ultra / Ultra 253B **[Browse all models →](/data/models)**
-🤖 AI Agents (50) — coding agents, frameworks, enterprise platforms +🤖 AI Agents (67) — coding agents, personal assistants, frameworks, platforms -**Coding & autonomous:** Claude Code, Claude Agent SDK, OpenAI Codex, Devin, Cursor, GitHub Copilot coding agent, Google Jules, Gemini CLI, Manus +**Coding & autonomous:** Claude Code, Claude Agent SDK, OpenAI Codex, Devin, Cursor, GitHub Copilot coding agent, Google Jules, Gemini CLI, Cline, OpenCode, Goose, Warp, JetBrains Junie, Factory Droids, Manus + +**Personal & workspace agents:** Claude Cowork, ChatGPT Agent, Microsoft Scout, OpenClaw, Perplexity Comet, Poke + +**App builders & agentic IDEs:** Replit Agent, Lovable, Google Antigravity, Amazon Kiro, ByteDance Trae **Frameworks:** OpenAI Agents SDK, Google ADK, Microsoft Agent Framework, AWS Strands, LangGraph, CrewAI, LlamaIndex, Pydantic AI, smolagents, Mastra, Dify @@ -155,11 +162,11 @@ Predefined profiles: `balanced` · `security_first` · `performance_focused` ·
-🔌 MCP Servers (46) — incl. security advisories on archived servers +🔌 MCP Servers (61) — incl. security advisories on archived servers -**Top ecosystem:** Context7, Chrome DevTools MCP, Playwright MCP, Serena +**Top ecosystem:** Context7, Chrome DevTools MCP, Playwright MCP, Serena, Exa, Browserbase -**Official vendor:** GitHub, Figma, Stripe, Notion, Vercel, Hugging Face, Zapier, Apify, Firecrawl, shadcn +**Official vendor:** GitHub, Figma, Stripe, PayPal, Shopify, Notion, Vercel, Snowflake, Databricks, Salesforce, HubSpot, Asana, Box, Grafana, Neon, Canva, ClickHouse, Neo4j, Hugging Face, Zapier, Apify, Firecrawl, shadcn **Reference:** the 7 actively maintained servers (fetch, git, filesystem, memory, time, sequential-thinking, everything) — plus the **archived** ones (Puppeteer, Postgres, SQLite, Slack, …) flagged with security advisories so you don't `npx` your way into a CVE diff --git a/app/agents/[id]/page.tsx b/app/agents/[id]/page.tsx index 7095a67..4bb09e4 100644 --- a/app/agents/[id]/page.tsx +++ b/app/agents/[id]/page.tsx @@ -1,5 +1,5 @@ import { getEntityById, getRelatedEntities, getEntitiesByType } from '@/lib/data'; -import { calculateOverallScore } from '@/framework/schema/types'; +import { calculateOverallScore, getScoreTheme } from '@/framework/schema/types'; import { ScoreBadge, ScoreBar } from '@/components/score-badge'; import { TrustVectorChart } from '@/components/trust-vector-chart'; import { formatDate } from '@/lib/utils'; @@ -62,14 +62,6 @@ export async function generateMetadata({ params }: { params: Promise<{ id: strin }; } -function getScoreClasses(score: number): { bg: string; text: string; label: string } { - if (score >= 90) return { bg: 'bg-emerald-500', text: 'text-white', label: 'Exceptional' }; - if (score >= 75) return { bg: 'bg-sky-500', text: 'text-white', label: 'Strong' }; - if (score >= 60) return { bg: 'bg-amber-400', text: 'text-black', label: 'Adequate' }; - if (score >= 40) return { bg: 'bg-orange-500', text: 'text-white', label: 'Concerning' }; - return { bg: 'bg-red-500', text: 'text-white', label: 'Poor' }; -} - export default async function AgentDetailPage({ params }: { params: Promise<{ id: string }> }) { const { id } = await params; const entity = getEntityById(id); @@ -79,7 +71,7 @@ export default async function AgentDetailPage({ params }: { params: Promise<{ id } const overallScore = calculateOverallScore(entity); - const scoreInfo = getScoreClasses(overallScore); + const scoreInfo = getScoreTheme(overallScore); const relatedEntities = getRelatedEntities(entity); const { trust_vector } = entity; @@ -94,55 +86,53 @@ export default async function AgentDetailPage({ params }: { params: Promise<{ id return (
{/* Breadcrumb */} -
+
{/* Hero Section */} -
+
-

{entity.name}

+

{entity.name}

v{entity.version}

{entity.provider}

Agent {entity.tags?.slice(0, 4).map((tag) => ( - {tag} + {tag} ))}
-
{overallScore}
+
{overallScore}
{scoreInfo.label}
-
-
+
+
About This Agent

{entity.description}

@@ -151,10 +141,10 @@ export default async function AgentDetailPage({ params }: { params: Promise<{ id
Last Evaluated:{' '} - {formatDate(entity.last_evaluated)} + {formatDate(entity.last_evaluated)}
{entity.website && ( - + Official Website )} @@ -169,26 +159,24 @@ export default async function AgentDetailPage({ params }: { params: Promise<{ id
-

Trust Vector Analysis

+

Trust Vector Analysis

-

Dimension Breakdown

+

Dimension Breakdown

{dimensions.map((dimension) => (
{dimension.icon} - {dimension.name} + {dimension.name}
@@ -196,18 +184,18 @@ export default async function AgentDetailPage({ params }: { params: Promise<{ id
-
+
{dimension.data.notes &&

{dimension.data.notes}

} {Object.entries(dimension.data.criteria).map(([key, criterion]) => ( -
+
- {key.replace(/_/g, ' ')} + {key.replace(/_/g, ' ')} {criterion.score !== undefined && }
{criterion.score !== undefined && }

{criterion.methodology}

-
-
Evidence
+
+
Evidence
{criterion.evidence.map((ev, idx) => (
{ev.source} @@ -216,7 +204,7 @@ export default async function AgentDetailPage({ params }: { params: Promise<{ id ))}
-
-
+
Strengths
    @@ -250,16 +237,15 @@ export default async function AgentDetailPage({ params }: { params: Promise<{ id
-
+
Limitations