Appearance
Provider Capabilities API
Overview
The Provider Capabilities API provides detailed information about AI provider availability, model specifications, performance metrics, and pricing. This API is used internally by the routing system and can be accessed by enterprise customers for capacity planning and provider evaluation.
Endpoint
GET /api/v1/providers/capabilities
GET /api/v1/providers/{providerId}/capabilitiesAuthentication
Requires valid API key or session authentication.
Rate Limits
- Free Tier: 100 requests/hour
- Standard Tier: 1,000 requests/hour
- Professional Tier: 10,000 requests/hour
- Enterprise Tier: Unlimited
Response Format
Get All Providers
GET /api/v1/providers/capabilitiesjson
{
"success": true,
"providers": [
{
"provider": "string",
"status": "healthy" | "degraded" | "unhealthy" | "maintenance",
"lastUpdated": "string (ISO 8601)",
"models": [
{
"id": "string",
"name": "string",
"type": "text" | "chat" | "embedding" | "image" | "multimodal",
"capabilities": {
"maxTokens": number,
"contextWindow": number,
"streaming": boolean,
"functionCalling": boolean,
"vision": boolean,
"codeExecution": boolean,
"fineTuning": boolean,
"plugins": boolean
},
"pricing": {
"inputTokens": number,
"outputTokens": number,
"flatRate": number,
"volumeDiscounts": [
{
"threshold": number,
"discount": number
}
]
},
"performance": {
"avgLatency": number,
"p95Latency": number,
"throughput": number,
"concurrencyLimit": number
},
"quality": {
"benchmarkScore": number,
"specialties": ["string"],
"languages": ["string"],
"limitations": ["string"]
},
"availability": {
"uptime": number,
"regions": ["string"],
"quotas": {
"requestsPerMinute": number,
"tokensPerDay": number,
"concurrent": number
}
}
}
],
"endpoints": [
{
"url": "string",
"region": "string",
"latency": number,
"available": boolean
}
],
"authentication": {
"type": "api_key" | "oauth" | "jwt",
"headerName": "string",
"format": "string"
},
"rateLimits": {
"global": {
"requestsPerSecond": number,
"requestsPerMinute": number,
"requestsPerHour": number
},
"perUser": {
"requestsPerMinute": number,
"tokensPerHour": number
}
},
"sla": {
"availability": number,
"support": "none" | "business_hours" | "24x7",
"responseTime": number
}
}
],
"metadata": {
"totalProviders": number,
"healthyProviders": number,
"lastUpdated": "string",
"cacheAge": number
}
}Get Specific Provider
GET /api/v1/providers/openai/capabilitiesjson
{
"success": true,
"provider": {
"provider": "openai",
"status": "healthy",
"lastUpdated": "2024-01-15T10:30:00Z",
"models": [
{
"id": "gpt-4-turbo",
"name": "GPT-4 Turbo",
"type": "chat",
"capabilities": {
"maxTokens": 128000,
"contextWindow": 128000,
"streaming": true,
"functionCalling": true,
"vision": true,
"codeExecution": false,
"fineTuning": false,
"plugins": false
},
"pricing": {
"inputTokens": 0.01,
"outputTokens": 0.03,
"flatRate": null,
"volumeDiscounts": [
{
"threshold": 1000000,
"discount": 0.1
}
]
},
"performance": {
"avgLatency": 500,
"p95Latency": 1500,
"throughput": 1000,
"concurrencyLimit": 100
},
"quality": {
"benchmarkScore": 95,
"specialties": ["reasoning", "creative", "analysis"],
"languages": ["en", "es", "fr", "de", "ja", "zh"],
"limitations": ["May hallucinate", "Training cutoff"]
},
"availability": {
"uptime": 0.999,
"regions": ["us-east", "us-west", "eu-west"],
"quotas": {
"requestsPerMinute": 10000,
"tokensPerDay": 10000000,
"concurrent": 100
}
}
}
],
"endpoints": [
{
"url": "https://api.openai.com/v1/chat/completions",
"region": "us-east",
"latency": 120,
"available": true
}
],
"authentication": {
"type": "api_key",
"headerName": "Authorization",
"format": "Bearer {token}"
},
"rateLimits": {
"global": {
"requestsPerSecond": 1000,
"requestsPerMinute": 10000,
"requestsPerHour": 1000000
},
"perUser": {
"requestsPerMinute": 500,
"tokensPerHour": 100000
}
},
"sla": {
"availability": 99.9,
"support": "24x7",
"responseTime": 4
}
}
}Provider Status Levels
| Status | Description | Impact |
|---|---|---|
| healthy | Normal operation | Full routing |
| degraded | Some issues, higher latency | Reduced routing priority |
| unhealthy | Significant issues | Minimal routing, fallback only |
| maintenance | Scheduled maintenance | No routing |
Model Types
| Type | Description | Use Cases |
|---|---|---|
| text | Text completion | Content generation, summarization |
| chat | Conversational AI | Chatbots, Q&A, dialogue |
| embedding | Vector embeddings | Search, similarity, clustering |
| image | Image processing | Vision tasks, image analysis |
| multimodal | Multiple modalities | Complex reasoning with images/text |
Model Capabilities
| Capability | Description |
|---|---|
maxTokens | Maximum tokens per request |
contextWindow | Maximum context length |
streaming | Real-time response streaming |
functionCalling | Structured function calls |
vision | Image understanding |
codeExecution | Code interpretation and execution |
fineTuning | Custom model training |
plugins | External tool integration |
Pricing Structure
Token-based Pricing
Most providers charge per token with separate input/output rates:
json
{
"pricing": {
"inputTokens": 0.01, // USD per 1K input tokens
"outputTokens": 0.03, // USD per 1K output tokens
"flatRate": null // Alternative: flat rate per request
}
}Volume Discounts
Enterprise customers may receive volume discounts:
json
{
"volumeDiscounts": [
{
"threshold": 1000000, // Tokens per month
"discount": 0.1 // 10% discount
},
{
"threshold": 10000000,
"discount": 0.2 // 20% discount
}
]
}Performance Metrics
| Metric | Description | Unit |
|---|---|---|
avgLatency | Average response time | milliseconds |
p95Latency | 95th percentile latency | milliseconds |
throughput | Tokens processed per second | tokens/sec |
concurrencyLimit | Max concurrent requests | requests |
Quality Assessment
Benchmark Scores
Quality scores are based on standardized benchmarks:
- 95-100: Excellent (GPT-4 class)
- 90-94: Very Good (Claude-3 class)
- 85-89: Good (GPT-3.5 class)
- 80-84: Fair (Smaller models)
- <80: Limited capability
Specialties
Models are tagged with their strengths:
reasoning: Logical reasoning and problem solvingcreative: Creative writing and ideationanalysis: Data analysis and interpretationcode: Programming and code generationmath: Mathematical problem solvingfactual: Factual knowledge and accuracy
Regional Availability
Supported Regions
| Region | Code | Description |
|---|---|---|
| US East | us-east | US East Coast (Virginia) |
| US West | us-west | US West Coast (California) |
| EU West | eu-west | Europe (Ireland/Frankfurt) |
| EU Central | eu-central | Europe (Frankfurt/Amsterdam) |
| Asia Pacific | ap-east | Asia Pacific (Singapore/Tokyo) |
| Asia East | asia-east | East Asia (Tokyo/Seoul) |
Data Residency
Some providers offer data residency guarantees:
json
{
"dataResidency": {
"us-east": {
"dataLocation": "United States",
"compliance": ["SOX", "HIPAA"]
},
"eu-west": {
"dataLocation": "European Union",
"compliance": ["GDPR"]
}
}
}Error Responses
404 Not Found
json
{
"error": "Provider not found",
"providerId": "invalid-provider"
}503 Service Unavailable
json
{
"error": "Provider capabilities temporarily unavailable",
"retryAfter": 300
}Example Usage
Compare Providers (JavaScript)
javascript
async function compareProviders(requirements) {
const response = await fetch('/api/v1/providers/capabilities');
const data = await response.json();
const suitable = data.providers.filter(provider => {
return provider.models.some(model => {
const meets =
model.capabilities.maxTokens >= requirements.minTokens &&
model.performance.avgLatency <= requirements.maxLatency &&
model.pricing.inputTokens <= requirements.maxCostPerToken;
return meets;
});
});
return suitable.sort((a, b) => {
// Sort by average cost (input + output)
const costA = a.models[0].pricing.inputTokens + a.models[0].pricing.outputTokens;
const costB = b.models[0].pricing.inputTokens + b.models[0].pricing.outputTokens;
return costA - costB;
});
}
// Usage
const requirements = {
minTokens: 32000,
maxLatency: 1000,
maxCostPerToken: 0.02
};
const providers = await compareProviders(requirements);
console.log('Suitable providers:', providers.map(p => p.provider));Monitor Provider Health (Python)
python
import requests
import time
def monitor_provider_health(providers, check_interval=300):
"""Monitor provider health every 5 minutes"""
while True:
for provider in providers:
try:
response = requests.get(f'/api/v1/providers/{provider}/capabilities')
data = response.json()
status = data['provider']['status']
uptime = data['provider']['sla']['availability']
if status != 'healthy':
print(f"⚠️ {provider}: {status}")
if uptime < 99.0:
print(f"📉 {provider}: {uptime}% uptime")
except Exception as e:
print(f"❌ {provider}: Error checking status - {e}")
time.sleep(check_interval)
# Monitor critical providers
monitor_provider_health(['openai', 'anthropic', 'deepseek'])Cost Calculator
javascript
class ProviderCostCalculator {
constructor(capabilities) {
this.capabilities = capabilities;
}
calculateCost(provider, model, inputTokens, outputTokens, volume = 0) {
const providerData = this.capabilities.find(p => p.provider === provider);
if (!providerData) throw new Error(`Provider ${provider} not found`);
const modelData = providerData.models.find(m => m.id === model);
if (!modelData) throw new Error(`Model ${model} not found`);
let inputCost = (inputTokens / 1000) * modelData.pricing.inputTokens;
let outputCost = (outputTokens / 1000) * modelData.pricing.outputTokens;
let totalCost = inputCost + outputCost;
// Apply volume discounts
if (volume > 0 && modelData.pricing.volumeDiscounts) {
for (const discount of modelData.pricing.volumeDiscounts) {
if (volume >= discount.threshold) {
totalCost *= (1 - discount.discount);
}
}
}
return {
inputCost,
outputCost,
totalCost,
costPer1KTokens: totalCost / ((inputTokens + outputTokens) / 1000)
};
}
findCheapestProvider(inputTokens, outputTokens, requirements = {}) {
const options = [];
for (const provider of this.capabilities) {
for (const model of provider.models) {
// Check requirements
if (requirements.minQuality && model.quality.benchmarkScore < requirements.minQuality) {
continue;
}
if (requirements.maxLatency && model.performance.avgLatency > requirements.maxLatency) {
continue;
}
const cost = this.calculateCost(provider.provider, model.id, inputTokens, outputTokens);
options.push({
provider: provider.provider,
model: model.id,
cost: cost.totalCost,
quality: model.quality.benchmarkScore,
latency: model.performance.avgLatency
});
}
}
return options.sort((a, b) => a.cost - b.cost);
}
}Best Practices
Caching
Provider capabilities are cached for 5 minutes. For real-time routing decisions, implement local caching:
javascript
class ProviderCapabilitiesCache {
constructor(ttl = 300000) { // 5 minutes
this.cache = new Map();
this.ttl = ttl;
}
async get(providerId = null) {
const key = providerId || 'all';
const cached = this.cache.get(key);
if (cached && Date.now() - cached.timestamp < this.ttl) {
return cached.data;
}
const url = providerId
? `/api/v1/providers/${providerId}/capabilities`
: '/api/v1/providers/capabilities';
const response = await fetch(url);
const data = await response.json();
this.cache.set(key, {
data,
timestamp: Date.now()
});
return data;
}
}Health Monitoring
Implement provider health monitoring for routing decisions:
python
def get_healthy_providers():
"""Get only healthy providers for routing"""
response = requests.get('/api/v1/providers/capabilities')
data = response.json()
return [
provider for provider in data['providers']
if provider['status'] == 'healthy' and
provider['sla']['availability'] > 99.0
]Changelog
Version 1.0.0 (Current)
- Initial release
- Support for 5 major providers
- Performance and pricing metrics
- Regional availability data
- Quality benchmarking
