Endpoints: 28,729MCP servers: 18,413Payout addresses: 2,071Paid calls: 1,545Letters: 14Defects: 1,324counted 4 min ago
teppi

Server definition

Hash
sha256:62b4ac86a8b9949636fbf8820cc651f16e3a0a955127cd423596b81a7d866129
What it is
What a remote MCP server returned when asked what it offers: 9 tools

The blob, as servednamed by its sha256

{ "instructions": "Cloud World Model Streamable HTTP MCP server. build_revision=665f8c62f2f8 build_timestamp=2026-10-04T01:22:11.986Z simulation_create_schema_fingerprint=sha256:304ef9dbb8811824c8c0a350e5626d6f230edb54e26f85b48c7738830166891a anonymous_mcp_capability_version=proxy-safe-v1", "tools": [ { "description": "Hydrate one built-in scenario from the live Cloud World Model scenario library. Prerequisite: a scenario id returned by scenario.list. Returns the complete selected scenario graph, including resources and connections plus optional seed, resilienceConfig, protectedResilienceConfig, traffic/failure presets, named traffic-phase summaries, activeFailurePhases and optionalFailurePhases (type, resource/zone target, severity, step range), retry-workload disclosure, and real-world incident metadata. The response includes both title and name for compatibility; pass resources and connections, and optionally seed/resilienceConfig, to simulation.create when you need to edit or inspect the graph. For the shorter handoff, pass the id as scenarioId instead. The likely next tool is simulation.create.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "scenarioId": { "description": "Scenario identifier returned by scenario.list", "minLength": 1, "type": "string" } }, "required": [ "scenarioId" ], "type": "object" }, "name": "scenario.get", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "properties": { "activeFailurePhases": { "items": { "additionalProperties": false, "properties": { "endStep": { "minimum": 0, "type": "number" }, "isActive": { "default": true, "type": "boolean" }, "name": { "type": "string" }, "severity": { "default": "moderate", "enum": [ "minor", "moderate", "severe" ], "type": "string" }, "startStep": { "minimum": 0, "type": "number" }, "targetProvider": { "enum": [ "aws", "gcp", "azure", "oci", "digitalocean" ], "type": "string" }, "targetRegion": { "type": "string" }, "targetResourceId": { "type": "string" }, "targetZone": { "type": "string" }, "type": { "enum": [ "instance_kill", "instance_down", "az_outage", "region_outage", "permanent_data_loss", "database_overload", "network_latency", "spot_interruption" ], "type": "string" } }, "required": [ "name", "type", "startStep" ], "type": "object" }, "type": "array" }, "activeTrafficPhases": { "items": { "additionalProperties": false, "properties": { "endStep": { "minimum": 0, "type": "number" }, "isActive": { "type": "boolean" }, "name": { "type": "string" }, "startStep": { "minimum": 0, "type": "number" }, "traffic": { "additionalProperties": false, "properties": { "endRps": { "minimum": 0, "type": "number" }, "peakRps": { "minimum": 0, "type": "number" }, "startRps": { "minimum": 0, "type": "number" } }, "type": "object" }, "type": { "enum": [ "ramp", "burst", "step", "wave", "custom" ], "type": "string" } }, "required": [ "name", "type", "startStep", "isActive", "traffic" ], "type": "object" }, "type": "array" }, "category": { "type": "string" }, "connections": { "description": "Full connection graph; pass to simulation.create", "items": { "additionalProperties": {}, "type": "object" }, "type": "array" }, "defaultFailureInjections": { "items": { "additionalProperties": {}, "type": "object" }, "type": "array" }, "defaultTrafficPatterns": { "items": { "additionalProperties": {}, "type": "object" }, "type": "array" }, "description": { "type": "string" }, "difficulty": { "type": "string" }, "duration": { "type": "string" }, "id": { "description": "Stable scenario identifier", "type": "string" }, "message": { "description": "Error or guidance message", "type": "string" }, "name": { "description": "Scenario display name; equivalent to title", "type": "string" }, "optionalFailurePhases": { "items": { "$ref": "#/properties/activeFailurePhases/items" }, "type": "array" }, "optionalTrafficPhases": { "items": { "$ref": "#/properties/activeTrafficPhases/items" }, "type": "array" }, "primaryPurpose": { "description": "Primary purpose of the scenario; absent means legacy purpose not specified", "enum": [ "educational", "chaos", "predictive", "optimization" ], "type": "string" }, "protectedResilienceConfig": { "additionalProperties": {}, "type": "object" }, "realWorldIncident": { "additionalProperties": {}, "type": "object" }, "resilienceConfig": { "additionalProperties": {}, "type": "object" }, "resources": { "description": "Full resource graph; pass to simulation.create", "items": { "additionalProperties": {}, "type": "object" }, "type": "array" }, "retryTrafficDisclosure": { "additionalProperties": false, "properties": { "dependencyCount": { "exclusiveMinimum": 0, "type": "integer" }, "enabled": { "const": true, "type": "boolean" }, "externalTraffic": { "type": "string" }, "internalRetryAttempts": { "type": "string" }, "maxConfiguredRetries": { "exclusiveMinimum": 0, "type": "integer" } }, "required": [ "enabled", "dependencyCount", "maxConfiguredRetries", "externalTraffic", "internalRetryAttempts" ], "type": "object" }, "seed": { "minimum": 0, "type": "integer" }, "status": { "description": "Result status; not_found when the requested scenario does not exist", "type": "string" }, "tags": { "items": { "type": "string" }, "type": "array" }, "title": { "description": "Scenario title", "type": "string" } }, "type": "object" } }, { "description": "List the built-in demo scenarios as compact catalog cards — stable IDs, title/name, description, difficulty, tags, category, duration, provider summary, resource/connection counts, named active/optional traffic phases, and retry-workload disclosure. Use it as the first call when you want a ready-made architecture instead of designing one; the cards intentionally omit resource, connection, traffic-pattern, and failure-injection graphs. Anonymous discovery includes only scenarios with at most 10 resources so every listed card is demo-creatable. No prerequisites. Optionally narrow discovery with provider, category, and/or difficulty filters; omit them to receive the complete demo-creatable catalog. Pass a returned id as scenarioId to simulation.create for server-side expansion, or pass it to scenario.get when you need to inspect the full graph. Larger scenarios require an authenticated session. Returns named activeFailurePhases and optionalFailurePhases with type, resource/zone target, severity, and step range. No API key required. The likely next tool is scenario.get.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "category": { "description": "Only scenarios in this category, such as scaling, failure, reliability, networking, or cost", "minLength": 1, "type": "string" }, "difficulty": { "description": "Only scenarios at this difficulty level", "enum": [ "beginner", "intermediate", "advanced" ], "type": "string" }, "provider": { "description": "Only scenarios that include resources from this cloud provider", "enum": [ "aws", "gcp", "azure", "oci", "digitalocean" ], "type": "string" } }, "type": "object" }, "name": "scenario.list", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "properties": { "scenarios": { "description": "Available demo scenarios", "items": { "additionalProperties": true, "properties": { "activeFailurePhases": { "description": "Scheduled active failures: name, type, target resource/zone, severity and step range; no parameters", "items": { "additionalProperties": false, "properties": { "endStep": { "minimum": 0, "type": "number" }, "isActive": { "default": true, "type": "boolean" }, "name": { "type": "string" }, "severity": { "default": "moderate", "enum": [ "minor", "moderate", "severe" ], "type": "string" }, "startStep": { "minimum": 0, "type": "number" }, "targetProvider": { "enum": [ "aws", "gcp", "azure", "oci", "digitalocean" ], "type": "string" }, "targetRegion": { "type": "string" }, "targetResourceId": { "type": "string" }, "targetZone": { "type": "string" }, "type": { "enum": [ "instance_kill", "instance_down", "az_outage", "region_outage", "permanent_data_loss", "database_overload", "network_latency", "spot_interruption" ], "type": "string" } }, "required": [ "name", "type", "startStep" ], "type": "object" }, "type": "array" }, "activeTrafficPhases": { "description": "Named active catalog traffic phases with type, step range, and workload shape", "items": { "additionalProperties": false, "properties": { "endStep": { "minimum": 0, "type": "number" }, "isActive": { "type": "boolean" }, "name": { "type": "string" }, "startStep": { "minimum": 0, "type": "number" }, "traffic": { "additionalProperties": false, "properties": { "endRps": { "minimum": 0, "type": "number" }, "peakRps": { "minimum": 0, "type": "number" }, "startRps": { "minimum": 0, "type": "number" } }, "type": "object" }, "type": { "enum": [ "ramp", "burst", "step", "wave", "custom" ], "type": "string" } }, "required": [ "name", "type", "startStep", "isActive", "traffic" ], "type": "object" }, "type": "array" }, "category": { "description": "Scenario category", "type": "string" }, "connectionCount": { "description": "Number of connections in the scenario graph", "type": "number" }, "description": { "description": "What this scenario demonstrates; capped at 500 characters with an ellipsis when truncated", "type": "string" }, "difficulty": { "description": "Scenario difficulty", "type": "string" }, "duration": { "description": "Expected scenario duration", "type": "string" }, "id": { "description": "Stable scenario identifier — pass to scenario.get", "type": "string" }, "name": { "description": "Scenario display name; equivalent to title", "type": "string" }, "optionalFailurePhases": { "description": "Disabled optional failure presets; not scheduled unless enabled", "items": { "$ref": "#/properties/scenarios/items/properties/activeFailurePhases/items" }, "type": "array" }, "optionalTrafficPhases": { "description": "Named optional catalog traffic phases; these are inactive until enabled in the workspace", "items": { "$ref": "#/properties/scenarios/items/properties/activeTrafficPhases/items" }, "type": "array" }, "primaryPurpose": { "description": "Primary purpose: educational guided scenario, chaos injected failure, predictive capacity simulation, or optimization configuration comparison", "enum": [ "educational", "chaos", "predictive", "optimization" ], "type": "string" }, "provider": { "description": "Primary cloud provider", "type": "string" }, "providerSummary": { "description": "Compact provider summary", "type": "string" }, "providers": { "description": "Cloud providers represented in the scenario", "items": { "type": "string" }, "type": "array" }, "resourceCount": { "description": "Number of resources in the scenario graph", "type": "number" }, "retryTrafficDisclosure": { "additionalProperties": false, "description": "When present, separates external catalog traffic from modeled internal retry attempts", "properties": { "dependencyCount": { "exclusiveMinimum": 0, "type": "integer" }, "enabled": { "const": true, "type": "boolean" }, "externalTraffic": { "type": "string" }, "internalRetryAttempts": { "type": "string" }, "maxConfiguredRetries": { "exclusiveMinimum": 0, "type": "integer" } }, "required": [ "enabled", "dependencyCount", "maxConfiguredRetries", "externalTraffic", "internalRetryAttempts" ], "type": "object" }, "revision": { "description": "Catalog revision; null when unavailable", "type": [ "string", "null" ] }, "tags": { "description": "Discovery tags", "items": { "type": "string" }, "type": "array" }, "title": { "description": "Scenario title", "type": "string" }, "version": { "description": "Catalog version; null when unavailable", "type": [ "string", "null" ] } }, "type": "object" }, "type": "array" } }, "required": [ "scenarios" ], "type": "object" } }, { "description": "Create a temporary anonymous demo cloud simulation from a list of resources and connections (max 2 active simulations per client, up to 10 resources; the returned simulationId is a short-lived unguessable capability that survives MCP transport teardown, but it is cleaned up when the demo lifetime expires or the simulation is deleted). No API key required for this temporary anonymous demo operation. Built-in scenario workflow: call `scenario.list` and pass a returned card's `id` as `scenarioId` to `simulation.create` for server-side graph expansion. For full control, call `scenario.get` and pass its hydrated `resources` and `connections` arrays instead. These are two alternatives — do not send `scenarioId` with `resources` or `connections`. For the catalog EKS Spot Interruption Migration scenario, you may set `scenarioOverrides: { eksSpotInterruption: { startupSeconds } }` with an integer startupSeconds from 0 through 3600 to test a different readiness deadline without copying the graph; this override requires scenarioId and is mutually exclusive with resources and connections. `scenario.list` returns graph-free cards with bounded active/optional traffic-phase and retry-workload summaries; it is not a source of resource, connection, traffic-pattern, or failure-injection graphs. Scenario traffic and failure presets are not applied automatically. Use it to start any simulation workflow — either with hydrated resources and connections from scenario.get or your own architecture. Do not use it to modify an existing simulation (use simulation.inject_traffic to change load). For the exact owned typical fit, set appWeight:'typical', location.regionKey:'us-east-2' on all four AWS nodes, one ALB with serviceFamily:'alb' and loadBalancerScheme:'internal', two m5.large compute apps with workload:'crud-typical', appRuntime:'node', appWorkerCount:2, appDbPoolSize:250, one db.r5.large MySQL with workloadDatabaseEngine:'mysql', workloadDatabaseVersion:'8.0', maxConnections:500, ALB→each app→DB connections, minInstances=maxInstances=2, autoscaling:false, traffic 20–300 RPS. Only typical-fit-20, typical-fit-100 and typical-fit-200 tuned the typical-v1-20260927c/6aa574d7ff9d3080b88b221bcd59f7d218ae37f0 fit; 300 is an independent holdout and 500 is diagnostic only. Other typical workloads are modeled, not owned. P99 has distinct per-percentile provenance: latencyP99Basis identifies the owned in-VPC internal-ALB fit, a scaled-from-fit estimate (not directly measured), or an uncalibrated generic model. On the exact healthy lean owned graph at 10–1,000 offered target RPS, P99=max(final P95, 7.021919127633514 + 0.00027013891327780484*T) ms; only 10/100/500 RPS were fit, 1,000 RPS was held out. Lean M5 scaling is not a new measurement; typical/heavy and active failures retain uncalibrated P99. predictionEvidence.latencyP99 has measured 0.65–1.35×, scaled 0.50–1.50× (beyond 1,000: 0.25–2×), or uncalibrated 0.50–2× (beyond: 0.25–3×) assumption bounds centered on final P99. These are not confidence intervals or provider measurements. Historical evidence may omit P99. latencyBasis describes the general modeled latency path; use latencyP99Basis specifically for P99. P99 is diagnostic, not scored. For compute, set characteristics.capacityRps for an explicit per-node RPS ceiling at which CPU reaches ~95%; do not use maxThroughput for that compute contract. Kubernetes rejects capacityRps: set maxThroughput for the total cluster RPS ceiling, or nodePools[].maxThroughput for per-node pool capacity. Omitted compute capacityRps uses the selected catalog tier and can intentionally produce a stressed baseline (for example, the AWS m5.large catalog denominator is 2,000 RPS); for a healthy, capacity-bounded compute experiment, declare an explicit per-node capacity such as 500 RPS. That value is an experiment control, not a universal hardware fact. For OCI flexible compute shapes, pass the documented positive integer characteristics.ocpus explicitly; VM.Standard.E4.Flex accepts 1–64 OCPUs and each OCPU maps to 2 vCPUs. OCPU count establishes capacity dimensions only, not provider-specific performance, throughput, or price. An uncounted flexible shape remains an unverified generic estimate. Check GET /api/prediction/generic-shapes for the catalog-derived generic fallback inventory. New prediction-only GCP standard capacity entries include e2-standard-2/4/8/16/32, n1-standard-1/2/4/8/16/32/64/96, and n2-standard-16/32/48/64/80/96/128; provider specifications establish vCPU/memory dimensions, not CWM performance or pricing. Version 1 predictionEvidence explicitly reports legacyGeneric at the top level and on every appCpuByResource item; its note identifies each generic resource's shape and fallback reason. For generic fixed compute, characteristics.instanceCount accepts integer 1–100 represented VMs; capacity aggregates and CPU is per VM. Do not combine it with autoscaling:true, minInstances, or maxInstances. Aurora Serverless v2 remains limited to 1 with multiAz:false or 2 with multiAz:true. For Aurora Serverless ACU limits, use characteristics.config.minCapacity/maxCapacity or flat characteristics.minCapacity/maxCapacity. The exact AWS database shape with serviceFamily: 'aurora-serverless' and size: 'db.serverless' also accepts flat characteristics.minAcu/maxAcu; those aliases are rejected elsewhere, including at the resource root or inside config. For that shape, multiAz:true with instanceCount:2 creates a separately billable reader (<writer-id>-reader) in another AZ. Inspect returned resources and metrics before using simulation.step to observe modeled failover; no AWS timing guarantee is implied. For database connection budgets, set characteristics.connectionDemand on a database: {mode:'declared',declaredConnections:240} uses that plan-time demand without RPS; {mode:'max',declaredConnections:240,idlePoolFloor:200} takes the maximum of load-derived demand and the declared/floor values; omitted configuration preserves load-derived behavior. declaredConnections and idlePoolFloor are ASSUMPTIONS / plan-time budgets (for example, replicas × per-pod pool size), not observed live DB connections. Set maxConnections to the usable limit you intend to test. Per-database metrics report connectionDemandMode, loadDerivedConnections, declaredConnections/idlePoolFloor, and modeledConnections; cost and DB CPU/latency remain based on existing load-driven behavior. Demand above the usable limit adds a bounded, rule-based pool-saturation error signal; it is not a provider-calibrated rate. Capacity, node-bound, SKU, and autoscaling values supplied through this MCP tool are recorded as agent-supplied in the immutable normalizationReceipt; request responseMode: 'full' to inspect it. Generic GKE telemetry and recovery apply only to worker nodes; the control-plane management fee is cost-only, with no modeled control-plane CPU, API throttling, or cooldown. To bound the autoscaled fleet size, set the top-level maxInstances / minInstances parameters. If you do not set maxInstances, the engine uses the provider default — AWS 50, GCP 15, Azure/OCI/DigitalOcean 10 — which may be much larger than your intended fleet size. The response includes effectiveMaxInstances / effectiveMinInstances so you can confirm the bounds that will be enforced. For a targeted CPU HPA scale-out threshold, send the canonical autoscalingTargetCpu field in this create call (for example, autoscalingTargetCpu: 70 for GKE). The compatible aliases scaleOutCpuThreshold, scaleOutCpuPercent, and autoscaleTargetCpuPercent are also accepted; if more than one is sent, their values must agree. Every create response includes hpaAudit with the supplied field, persisted thresholds, and any provider default. For ECS Fargate CPU-only target tracking, set ecsCpuTargetTracking: true, autoscalingTargetCpu, minInstances/maxInstances, and optional scaleOutCooldownSeconds/scaleInCooldownSeconds with simulationSecondsPerStep (default 1). Inspect autoscalingConfig in the compact response or applicationAutoscalingPolicy in the full response. Latency and throughput do not trigger ECS scaling in this mode. These four TOP-LEVEL fields are simulation-wide — the engine applies one CPU threshold identically to every resource's scale decision by default. To make ONE resource scale at a different CPU target than the rest of the simulation (e.g. a GKE cluster scaling out at 60% while an EC2 fleet in the same simulation scales out at 80%), set characteristics.scaleOutCpuThreshold and/or characteristics.scaleInCpuThreshold on that specific resource instead — the per-resource value wins over the simulation-wide default for that resource only. A misnamed near-miss field nested under characteristics (e.g. targetCPUUtilizationPercentage) is rejected with a 400 explaining the correct field name — it is never silently dropped and defaulted. Responses are compact by default: id, name, status, traffic, and a per-resource summary (id, name, status, cpuPercent, routedRps, availabilityState, isRoutable, and recoveryBlockedReason when provided). Pass responseMode: 'full' to get the complete simulation object instead. During a failure workflow, lower traffic to serviceable levels before calling simulation.recover_resource, then use simulation.step until the recovered resource is healthy. Recovery progress is included when applicable: recoveryProgress.state is parked, cooling_down, or healthy, and its parkWindow/cooldown objects report totalSteps, completedSteps, remainingSteps, target, and requiredSteps. Poll simulation.step until state is healthy, then use simulation.metrics to inspect the resulting state and metrics. No prerequisites. Returns the created simulation's id, which every other simulation.* tool consumes; the new simulation also becomes this session's current simulation, so subsequent per-simulation tools may omit simulationId. The likely next tool is simulation.step to advance time. Do not call api.spec to learn the simulation workflow — the tool descriptions in this session contain everything needed. Authenticate with an API key for unlimited persistent simulations.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "appWeight": { "description": "Immutable root-level workload weight; defaults to typical with appWeightDefaulted=true. Lean is measured only for the eligible owned graph; heavy is an unsupported assumption. predictionEvidence reports per-quantity provenance and assumption ranges, not confidence intervals.", "enum": [ "lean", "typical", "heavy" ], "type": "string" }, "autoscaleTargetCpuPercent": { "description": "Grok-compatible alias for the CPU HPA scale-out target; if multiple target names are sent they must match.", "maximum": 100, "minimum": 0, "type": "number" }, "autoscalingTargetCpu": { "description": "Canonical CPU HPA scale-out target percent. CWM synthesizes unrelated autoscaling defaults.", "maximum": 100, "minimum": 0, "type": "number" }, "connections": { "description": "Directed connections between resources. For a built-in scenario, pass the hydrated connections from scenario.get; scenario.list cards are graph-free. Directed edges should describe traffic flow between resources; omit when using scenarioId.", "items": { "additionalProperties": false, "properties": { "label": { "description": "Optional label describing the connection type", "type": "string" }, "sourceId": { "description": "ID of the source (upstream) resource", "type": "string" }, "targetId": { "description": "ID of the target (downstream) resource", "type": "string" } }, "required": [ "sourceId", "targetId" ], "type": "object" }, "type": "array" }, "description": { "description": "Optional description of the simulation's purpose", "maxLength": 500, "type": "string" }, "ecsCpuTargetTracking": { "description": "Opt into CPU-only ECS Fargate target tracking.", "type": "boolean" }, "maxInstances": { "description": "Hard ceiling on the autoscaled compute fleet size, stored as autoscalingConfig.maxInstances. If omitted, the provider default applies (AWS 50, GCP 15, Azure/OCI/DigitalOcean 10) — which may be much larger than your intended fleet size.", "minimum": 1, "type": "integer" }, "minInstances": { "description": "Floor on the autoscaled compute fleet size, stored as autoscalingConfig.minInstances.", "minimum": 1, "type": "integer" }, "name": { "description": "Human-readable name for the simulation", "maxLength": 120, "type": "string" }, "resilienceConfig": { "additionalProperties": false, "description": "Optional retry/cascade resilience model returned by scenario.get", "properties": { "dependencies": { "description": "Dependency edges with retry and protection policies", "items": { "additionalProperties": false, "properties": { "authDependencyId": { "description": "Dependency receiving generated auth/token traffic", "maxLength": 128, "minLength": 1, "type": "string" }, "authRequestsPerAttempt": { "description": "Auth/token requests generated per dependency attempt", "maximum": 10, "minimum": 0, "type": "number" }, "capacity": { "additionalProperties": false, "properties": { "maxConcurrent": { "exclusiveMinimum": 0, "maximum": 1000000, "type": "integer" }, "maxRps": { "exclusiveMinimum": 0, "maximum": 500000, "type": "number" }, "meanServiceTimeMs": { "exclusiveMinimum": 0, "maximum": 120000, "type": "number" } }, "type": "object" }, "id": { "description": "Unique dependency edge ID", "maxLength": 128, "minLength": 1, "type": "string" }, "protection": { "additionalProperties": false, "properties": { "circuitBreaker": { "additionalProperties": false, "properties": { "enabled": { "type": "boolean" }, "failureRateThreshold": { "maximum": 1, "minimum": 0, "type": "number" }, "halfOpenMaxRequests": { "maximum": 10000, "minimum": 1, "type": "integer" }, "minimumRequests": { "maximum": 100000, "minimum": 1, "type": "integer" }, "openSteps": { "maximum": 120, "minimum": 1, "type": "integer" } }, "type": "object" }, "loadShedding": { "type": "boolean" }, "rateLimitRps": { "exclusiveMinimum": 0, "maximum": 500000, "type": "number" } }, "type": "object" }, "requestRatio": { "description": "Requests to target per original request", "maximum": 20, "minimum": 0, "type": "number" }, "retryPolicy": { "additionalProperties": false, "properties": { "backoffMs": { "description": "Initial retry backoff in milliseconds (default 100).", "maximum": 60000, "minimum": 0, "type": "integer" }, "backoffMultiplier": { "description": "Exponential backoff multiplier (default 2).", "maximum": 10, "minimum": 1, "type": "number" }, "jitterRatio": { "description": "Fractional backoff jitter from 0 to 1 (default 0.1).", "maximum": 1, "minimum": 0, "type": "number" }, "maxRetries": { "description": "Maximum retries for this dependency edge (0 disables retries; default 2).", "maximum": 8, "minimum": 0, "type": "integer" }, "retryActorId": { "description": "Actor producing retries, such as a gateway or client", "maxLength": 128, "minLength": 1, "type": "string" }, "retryBudgetRatio": { "description": "Retry RPS budget as a multiple of original edge RPS (default 2).", "maximum": 10, "minimum": 0, "type": "number" }, "retryBudgetRps": { "description": "Optional absolute retry RPS cap, also bounded by retryBudgetRatio.", "maximum": 500000, "minimum": 0, "type": "number" }, "timeoutMs": { "description": "Per-attempt client deadline in milliseconds (default 2000); includes queue wait, service, and latency.", "maximum": 120000, "minimum": 1, "type": "integer" } }, "type": "object" }, "sourceId": { "description": "Upstream resource ID", "minLength": 1, "type": "string" }, "targetId": { "description": "Downstream resource ID", "minLength": 1, "type": "string" } }, "required": [ "id", "sourceId", "targetId" ], "type": "object" }, "maxItems": 64, "type": "array" }, "enabled": { "description": "Master switch for retry/cascade modeling", "type": "boolean" }, "maxCascadeDepth": { "maximum": 8, "minimum": 1, "type": "integer" }, "maxGeneratedRps": { "exclusiveMinimum": 0, "maximum": 500000, "type": "number" }, "maxStepWork": { "maximum": 2048, "minimum": 1, "type": "integer" }, "retryGeneratedTrafficAffectsCost": { "type": "boolean" }, "scalingPolicies": { "description": "Capacity-observation policies, including intentional autoscaling blind spots", "items": { "additionalProperties": false, "properties": { "constrainedMetric": { "enum": [ "rps", "concurrency" ], "type": "string" }, "dependencyId": { "maxLength": 128, "minLength": 1, "type": "string" }, "id": { "maxLength": 128, "minLength": 1, "type": "string" }, "observationMetric": { "enum": [ "source_cpu", "capacity_utilization" ], "type": "string" }, "observedResourceId": { "minLength": 1, "type": "string" }, "scaleOutCapacityMultiplier": { "maximum": 20, "minimum": 1, "type": "number" }, "scaleOutThresholdPercent": { "maximum": 100, "minimum": 1, "type": "number" } }, "required": [ "id", "dependencyId", "constrainedMetric", "observationMetric", "scaleOutThresholdPercent" ], "type": "object" }, "maxItems": 32, "type": "array" }, "scheduledFaults": { "description": "Scheduled capacity, error, latency, concurrency, or traffic-surge faults", "items": { "additionalProperties": false, "properties": { "addedLatencyMs": { "maximum": 120000, "minimum": 0, "type": "number" }, "capacityPercent": { "maximum": 100, "minimum": 0, "type": "number" }, "dependencyId": { "type": "string" }, "endStep": { "minimum": 1, "type": "integer" }, "errorRate": { "maximum": 1, "minimum": 0, "type": "number" }, "id": { "maxLength": 128, "minLength": 1, "type": "string" }, "maxConcurrent": { "exclusiveMinimum": 0, "maximum": 1000000, "type": "integer" }, "startStep": { "minimum": 0, "type": "integer" }, "targetResourceId": { "type": "string" }, "trafficMultiplier": { "description": "traffic_surge demand multiplier", "exclusiveMinimum": 1, "maximum": 20, "type": "number" }, "type": { "description": "Fault type; traffic_surge increases root client demand", "enum": [ "capacity_limit", "concurrency_limit", "latency", "error_rate", "traffic_surge" ], "type": "string" } }, "required": [ "id", "type", "startStep" ], "type": "object" }, "maxItems": 32, "type": "array" }, "version": { "const": 1, "description": "Resilience model version", "type": "number" } }, "type": "object" }, "resources": { "description": "List of cloud resources composing this simulation. For a built-in scenario, pass the hydrated resources from scenario.get; scenario.list cards are graph-free. (max 10 in demo mode; mutually exclusive with scenarioId)", "items": { "additionalProperties": false, "properties": { "characteristics": { "additionalProperties": true, "description": "Capacity and sizing characteristics for the resource (extra keys such as costMultiplier pass through unchanged)", "properties": { "appDbPoolSize": { "description": "Owned typical: pool size 250 on each app.", "exclusiveMinimum": 0, "type": "integer" }, "appRuntime": { "description": "Owned typical app runtime: node.", "type": "string" }, "appWorkerCount": { "description": "Owned typical: 2 workers on each app.", "exclusiveMinimum": 0, "type": "integer" }, "autoscaling": { "description": "Mark this compute resource as the autoscaling primary target", "type": "boolean" }, "billingState": { "description": "Billing state: 'stopped' bills storage only, 'idle'/'detached' bill flat idle rates, 'deleted' bills nothing. Default 'active'.", "enum": [ "active", "idle", "stopped", "detached", "deleted" ], "type": "string" }, "capacityGB": { "description": "Storage capacity in GB — drives per-GB snapshot/backup idle cost and stopped-instance storage cost", "type": "number" }, "capacityRps": { "description": "Compute only: literal per-node RPS ceiling at which CPU reaches ~95%. Kubernetes rejects capacityRps; use maxThroughput for its total cluster ceiling.", "exclusiveMinimum": 0, "type": "number" }, "config": { "additionalProperties": true, "description": "Aurora Serverless ACU bounds. Nested config uses minCapacity/maxCapacity; flat characteristics.minAcu/maxAcu aliases are accepted only on the exact AWS Aurora Serverless v2 db.serverless shape.", "properties": { "maxCapacity": { "exclusiveMinimum": 0, "type": "number" }, "minCapacity": { "exclusiveMinimum": 0, "type": "number" } }, "type": "object" }, "connectionDemand": { "additionalProperties": false, "description": "Database only. Omit to preserve legacy load-derived connection use. `declared` and `max` require declaredConnections and/or idlePoolFloor. Demand above usable maxConnections adds a bounded rule-based pool-saturation error signal, not a provider-calibrated rate.", "properties": { "declaredConnections": { "description": "ASSUMPTION / plan-time peak pool budget (for example, replicas × per-pod pool size), not an observed live connection count.", "maximum": 1000000, "minimum": 0, "type": "integer" }, "idlePoolFloor": { "description": "ASSUMPTION / plan-time minimum idle pool footprint, not an observed live connection count.", "maximum": 1000000, "minimum": 0, "type": "integer" }, "mode": { "description": "load-derived keeps the legacy traffic estimate; declared uses only the plan-time declaredConnections/idlePoolFloor; max uses the greatest of load-derived and declared/floor demand.", "enum": [ "load-derived", "declared", "max" ], "type": "string" } }, "required": [ "mode" ], "type": "object" }, "instanceCount": { "description": "Generic fixed compute: represented VM count (integer 1–100); capacity aggregates and CPU is per VM. Cannot combine with autoscaling:true, minInstances, or maxInstances. AWS Aurora Serverless v2 remains 1 with multiAz:false or 2 with multiAz:true.", "maximum": 100, "minimum": 1, "type": "integer" }, "loadBalancerScheme": { "description": "Owned typical ALB scheme: internal.", "enum": [ "internal", "internet-facing" ], "type": "string" }, "maxAcu": { "description": "Flat maximum ACU alias only for AWS Aurora Serverless v2 size db.serverless; conflicts with maxCapacity are rejected.", "exclusiveMinimum": 0, "type": "number" }, "maxCapacity": { "description": "Aurora Serverless maximum ACU (flat form; nested config wins).", "exclusiveMinimum": 0, "type": "number" }, "maxConnections": { "description": "Fixed concurrent DB connection budget; Aurora Serverless defaults from maximum configured ACU.", "exclusiveMinimum": 0, "maximum": 1000000, "type": "integer" }, "maxThroughput": { "description": "Kubernetes: total cluster RPS ceiling; compute: legacy internal throughput scaling parameter (prefer capacityRps for compute).", "type": "number" }, "minAcu": { "description": "Flat minimum ACU alias only for AWS Aurora Serverless v2 size db.serverless; conflicts with minCapacity are rejected.", "exclusiveMinimum": 0, "type": "number" }, "minCapacity": { "description": "Aurora Serverless minimum ACU (flat form; nested config wins).", "exclusiveMinimum": 0, "type": "number" }, "multiAz": { "description": "AWS Aurora Serverless v2: true with instanceCount:2 creates a billable reader in a second AZ.", "type": "boolean" }, "ocpus": { "description": "OCI flexible compute shape capacity. Family-specific limits are validated against the provider shape; VM.Standard.E4.Flex accepts 1–64 OCPUs and 1 OCPU = 2 vCPUs. Not a performance or price claim.", "exclusiveMinimum": 0, "maximum": 126, "type": "integer" }, "scaleInCpuThreshold": { "description": "Compute or Kubernetes only — overrides the simulation-wide CPU HPA scale-in target for THIS resource's scale decisions only; every other resource keeps the simulation-wide default.", "maximum": 100, "minimum": 0, "type": "number" }, "scaleOutCpuThreshold": { "description": "Compute or Kubernetes only — overrides the simulation-wide CPU HPA scale-out target (see the top-level autoscalingTargetCpu) for THIS resource's scale decisions only; every other resource keeps the simulation-wide default.", "maximum": 100, "minimum": 0, "type": "number" }, "serviceFamily": { "description": "Idle-billed service family (e.g. 'nat-gateway', 'vpc-endpoint', 'eip-detached', 'ebs-snapshot') — determines the flat idle rate", "type": "string" }, "sessionAffinity": { "additionalProperties": false, "description": "Opt-in modeled sticky owners for generic compute; Kubernetes requires explicit workload replicas independent of nodes. Use identical fleetId/config on compute peers. Existing sessions do not migrate on scale-out; owner loss disconnects them. reconnect:'next-step' explicitly enables rebinding; default none. Session arrivals/lifetime are assumptions, not measured OpenShell data. Read full step metrics.sessionAffinity for per-owner load, rejects and disconnects.", "properties": { "arrivalsPerStep": { "default": 0, "minimum": 0, "type": "number" }, "fleetId": { "maxLength": 128, "minLength": 1, "type": "string" }, "initialSessions": { "default": 100, "minimum": 0, "type": "number" }, "maxSessionsPerOwner": { "default": 1000, "exclusiveMinimum": 0, "type": "number" }, "meanLifetimeSteps": { "default": 300, "minimum": 1, "type": "number" }, "mode": { "const": "sticky", "type": "string" }, "reconnect": { "default": "none", "enum": [ "none", "next-step" ], "type": "string" }, "requestsPerSessionPerSecond": { "default": 1, "exclusiveMinimum": 0, "type": "number" }, "workload": { "additionalProperties": false, "properties": { "autoscaling": { "default": true, "type": "boolean" }, "capacityRpsPerReplica": { "exclusiveMinimum": 0, "type": "number" }, "maxReplicas": { "maximum": 10000, "minimum": 1, "type": "integer" }, "minReplicas": { "maximum": 10000, "minimum": 1, "type": "integer" }, "replicas": { "maximum": 10000, "minimum": 1, "type": "integer" }, "targetCpuPercent": { "default": 70, "maximum": 100, "minimum": 1, "type": "number" } }, "required": [ "replicas", "minReplicas", "maxReplicas", "capacityRpsPerReplica" ], "type": "object" } }, "required": [ "mode", "fleetId" ], "type": "object" }, "size": { "description": "Instance or SKU size (e.g. 't3.medium', 'db.r5.large')", "type": "string" }, "workload": { "description": "Explicit 'crud-typical' on each owned typical app; 'crud' selects lean identity on the exact lean graph.", "enum": [ "crud", "crud-typical" ], "type": "string" }, "workloadDatabaseEngine": { "description": "MySQL workload identity on the db.r5.large database.", "enum": [ "mysql" ], "type": "string" }, "workloadDatabaseVersion": { "description": "Owned typical MySQL workload version: 8.0.", "type": "string" } }, "type": "object" }, "id": { "description": "Unique identifier for this resource within the simulation", "type": "string" }, "location": { "additionalProperties": false, "description": "Resource region; set location.regionKey:'us-east-2' on each owned typical node (normalized to use2).", "properties": { "faultDomainKey": { "type": "string" }, "localityType": { "enum": [ "az", "zone", "availability_domain", "fault_domain" ], "type": "string" }, "providerLabel": { "type": "string" }, "regionKey": { "type": "string" }, "zoneKey": { "type": "string" } }, "required": [ "regionKey" ], "type": "object" }, "maxAcu": { "description": "Not accepted at resource level; use characteristics.maxAcu for the exact AWS Aurora Serverless v2 db.serverless shape." }, "minAcu": { "description": "Not accepted at resource level; use characteristics.minAcu for the exact AWS Aurora Serverless v2 db.serverless shape." }, "name": { "description": "Display name for the resource", "type": "string" }, "provider": { "default": "aws", "description": "Cloud provider hosting this resource", "enum": [ "aws", "gcp", "azure", "oci", "digitalocean" ], "type": "string" }, "recoveryPolicy": { "additionalProperties": false, "description": "Optional recovery thresholds and cooldown lengths for this resource", "properties": { "criticalCpuThreshold": { "maximum": 100, "minimum": 0, "type": "number" }, "criticalSteps": { "minimum": 1, "type": "integer" }, "failureParkSteps": { "minimum": 1, "type": "integer" }, "warningCpuThreshold": { "maximum": 100, "minimum": 0, "type": "number" }, "warningSteps": { "minimum": 1, "type": "integer" } }, "type": "object" }, "type": { "description": "Resource category", "enum": [ "compute", "database", "storage", "network", "cache", "queue", "kubernetes" ], "type": "string" } }, "required": [ "id", "type", "name" ], "type": "object" }, "type": "array" }, "responseMode": { "default": "compact", "description": "Response detail level. 'compact' (default) returns id, name, status, traffic, and a per-resource summary (id, name, status, cpuPercent) — keeps the response small for agent loops. 'full' returns the complete simulation object including all resource characteristics and connections.", "enum": [ "compact", "full" ], "type": "string" }, "scaleInCooldownSeconds": { "description": "ECS CPU target-tracking scale-in cooldown in simulated seconds.", "maximum": 86400, "minimum": 0, "type": "integer" }, "scaleOutCooldownSeconds": { "description": "ECS CPU target-tracking scale-out cooldown in simulated seconds.", "maximum": 86400, "minimum": 0, "type": "integer" }, "scaleOutCpuPercent": { "description": "Grok-compatible alias for the CPU HPA scale-out target; if multiple target names are sent they must match.", "maximum": 100, "minimum": 0, "type": "number" }, "scaleOutCpuThreshold": { "description": "Equivalent alias for autoscalingTargetCpu; if both are sent they must match.", "maximum": 100, "minimum": 0, "type": "number" }, "scenarioId": { "description": "Live scenario identifier from scenario.list; mutually exclusive with resources and connections", "minLength": 1, "type": "string" }, "seed": { "description": "Deterministic RNG seed for reproducible replays", "minimum": 0, "type": "integer" }, "simulationSecondsPerStep": { "description": "Simulated seconds per step for ECS cooldowns (default 1).", "exclusiveMinimum": 0, "maximum": 60, "type": "number" }, "traffic": { "default": 0, "description": "Initial traffic in requests per second (RPS)", "type": "number" } }, "required": [ "name" ], "type": "object" }, "name": "simulation.create", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "description": "Compact created-simulation summary by default (id, name, status, traffic, per-resource summary), the complete simulation object with responseMode 'full', or a structured limit response (status: limit_reached) when a demo cap is reached", "properties": { "appWeight": { "enum": [ "lean", "typical", "heavy" ], "type": "string" }, "appWeightDefaulted": { "type": "boolean" }, "autoscalingConfig": { "additionalProperties": true, "description": "Effective simulation scaling config; full response also contains the ECS resource's linked CPU-only target-tracking policy.", "properties": {}, "type": "object" }, "calibrationEvidence": { "additionalProperties": false, "description": "Owned-versus-modeled evidence and latency boundary.", "properties": { "calibrationId": { "description": "Versioned owned calibration identifier; present only when that calibration applies.", "type": "string" }, "kind": { "description": "owned for the exact AWS CRUD fit; modeled when the gate fails or another generic model applies.", "enum": [ "owned", "modeled" ], "type": "string" }, "latencyBasis": { "description": "General modeled latency path, such as in-VPC ALB rather than end-to-end; see latencyP99Basis for percentile-specific P99 provenance.", "type": "string" }, "latencyP99Basis": { "description": "Percentile-specific P99 basis: owned in-VPC internal-ALB fit, scaled from that fit (not directly measured), or uncalibrated generic model.", "type": "string" }, "note": { "description": "Names the fit scope and limitations. For canonical workload inference, states it is a modeling assumption. For modeled fallback, names all actual failed checks in stable order with resource IDs and safe expected/actual values; never only a generic 'not applied'.", "type": "string" } }, "required": [ "kind", "latencyBasis", "note" ], "type": "object" }, "effectiveConfigHash": { "description": "Versioned prediction hash over replay startup inputs, engine version, and calibration identity; see replayIdentity.effectiveConfigHash for the original replay-only hash.", "maxLength": 64, "minLength": 64, "type": "string" }, "effectiveMaxInstances": { "description": "The fleet-size ceiling the engine will enforce (autoscalingConfig.maxInstances, or the provider default when unset)", "type": "number" }, "effectiveMinInstances": { "description": "The fleet-size floor the engine will enforce (autoscalingConfig.minInstances, or the provider default when unset)", "type": "number" }, "engineVersion": { "description": "Simulation engine version used for this prediction.", "type": "string" }, "hpaAudit": { "additionalProperties": false, "description": "CPU HPA create-time audit: whether a target arrived, its accepted field, the persisted thresholds, and the provider-default explanation when omitted.", "properties": { "defaultExplanation": { "type": "string" }, "defaulted": { "type": "boolean" }, "effectiveScaleInCpuPercent": { "type": "number" }, "effectiveScaleOutCpuPercent": { "type": "number" }, "provider": { "type": "string" }, "requestedTargetCpu": { "type": [ "number", "null" ] }, "resourceCategory": { "type": "string" }, "suppliedField": { "type": [ "string", "null" ] }, "targetSupplied": { "type": "boolean" } }, "required": [ "targetSupplied", "suppliedField", "requestedTargetCpu", "effectiveScaleOutCpuPercent", "effectiveScaleInCpuPercent", "defaulted", "provider", "resourceCategory" ], "type": "object" }, "id": { "description": "Unique simulation ID — use with simulation.step, simulation.metrics, etc.", "type": "string" }, "name": { "description": "Simulation name", "type": "string" }, "normalizedConfig": { "additionalProperties": true, "description": "Engine-resolved billing parameters for every resource: cost multipliers, hourly rates, autoscale thresholds (scaleOut/scaleIn CPU %), GPU SKU, per-node token throughput, billing floor, connection limits. Use this immediately after create to verify the simulation was set up as intended — e.g. confirm which GPU SKU was resolved, the effective billing floor, or the autoscale CPU threshold that will drive scale-out.", "properties": { "resources": { "description": "Per-resource billing parameters resolved by the engine at create time", "items": { "additionalProperties": true, "properties": { "autoscaleThreshold": { "additionalProperties": false, "description": "Effective autoscale thresholds (provider profile merged with any explicit config)", "properties": { "scaleInCpuPercent": { "description": "CPU % below which a scale-in event fires", "type": "number" }, "scaleOutCpuPercent": { "description": "CPU % that triggers a scale-out event", "type": "number" } }, "required": [ "scaleOutCpuPercent", "scaleInCpuPercent" ], "type": "object" }, "behaviorModel": { "additionalProperties": true, "description": "Latency simulation model and GPU interconnect topology parameters for this resource. latencyPath identifies which engine path runs at step time. Two distinct model components: (1) throughput scaling — topologyThroughputFactor (resolveTopologyScalingFactor) bounds effective token capacity; topologyThroughputCalibrationStatus and topologyIsLegacyBaseline qualify that factor. (2) latency shape — topologyTtftFactor + topologyDecodeFactor (resolveTopologyLatencyFactors) scale TTFT and decode curves; topologyCalibrationStatus qualifies those factors. Both components are present on inferenceMode Kubernetes clusters; latencyPath is present on all resource types.", "properties": { "acceleratorResolution": { "description": "'catalog' = accelerator matched ACCELERATOR_TOKENS_PER_SEC; 'fallback' = unrecognised, defaults substituted (TTFT 600 ms, decode 700 tok/s, perNodeTokens 500)", "enum": [ "catalog", "fallback" ], "type": "string" }, "behaviourFidelity": { "description": "'modeled' = parameters from the performance catalog; 'estimated' = defaults substituted", "enum": [ "modeled", "estimated" ], "type": "string" }, "description": { "description": "Human-readable description of the latency model applied to this resource", "type": "string" }, "latencyPath": { "description": "Canonical latency simulation path identifier, e.g. 'gpu-inference/ttft-decode' or 'generic-kubernetes/load-saturation'. Machine-match against LATENCY_PATHS constants — do not string-parse.", "type": "string" }, "modelingNote": { "description": "Provenance disclaimer for latency figures. On inferenceMode clusters states that TTFT/P95/P99 are CWM simulation-model estimates from accelerator catalog parameters, not externally measured benchmarks. When topology is omitted also notes the legacy-baseline assumption (topologyThroughputFactor=1.0).", "type": "string" }, "topologyCalibrationStatus": { "description": "Calibration status of the TTFT+decode latency factors (resolveTopologyLatencyFactors). Entirely separate from topologyThroughputCalibrationStatus, which covers throughput scaling.", "enum": [ "calibrated", "estimated", "not_applicable" ], "type": "string" }, "topologyDecodeFactor": { "description": "Decode-rate latency scaling factor applied by the engine's GPU inference model (resolveTopologyLatencyFactors). Describes the TTFT+decode latency dimension — distinct from the throughput scaling factor.", "type": "number" }, "topologyInterNode": { "description": "Raw characteristics.topology.interNode value or null when absent", "type": [ "string", "null" ] }, "topologyIntraNode": { "description": "Raw characteristics.topology.intraNode value ('pcie', 'nvlink', 'nvlink-nvswitch', 'infiniband') or null when absent", "type": [ "string", "null" ] }, "topologyIsLegacyBaseline": { "description": "true when topology.intraNode was absent or unrecognised, meaning topologyThroughputFactor=1.0 reflects the pre-topology legacy baseline (not a fabric-specific measurement). false when an explicit, recognised intraNode value was supplied and the throughput factor is topology-modelled.", "type": "boolean" }, "topologyNodeCount": { "description": "Node count used when resolving topology latency factors (≥ 1)", "type": "number" }, "topologyThroughputCalibrationStatus": { "description": "Calibration status of the throughput scaling factor (resolveTopologyScalingFactor). Distinct from topologyCalibrationStatus, which covers TTFT+decode latency factors. 'estimated' for all current intraNode coefficients (engineering assumptions from bandwidth arithmetic). 'not_applicable' on non-inferenceMode resources. Reserve 'calibrated' for when a directly measured per-request throughput ratio is added.", "enum": [ "calibrated", "estimated", "not_applicable" ], "type": "string" }, "topologyThroughputFactor": { "description": "Throughput scaling coefficient applied by resolveTopologyScalingFactor to this cluster's token capacity (nodes × perNodeTokensPerSec × factor × gpuUtil/100). 1.0 when topology.intraNode is absent or unrecognised (legacy baseline, topologyIsLegacyBaseline=true). Known fabric penalties: pcie≈0.70, nvlink≈0.85, nvlink-nvswitch≈0.92, infiniband≈1.0. Entirely separate from topologyTtftFactor/topologyDecodeFactor, which describe the TTFT+decode latency model.", "type": "number" }, "topologyTtftFactor": { "description": "TTFT latency scaling factor applied by the engine's GPU inference model (resolveTopologyLatencyFactors). Describes the TTFT+decode latency dimension — distinct from the throughput scaling factor.", "type": "number" } }, "type": "object" }, "billingFloorCostPerHour": { "description": "Minimum cluster cost in USD/hr (cost at billingFloorNodes)", "type": "number" }, "billingFloorNodes": { "description": "Minimum node count the engine will ever bill — scale-in cannot go below this", "type": "number" }, "capacityProvenance": { "description": "Source of the effective compute capacity: explicit caller field, provider catalog, or generic fallback", "enum": [ "absent", "capacityRps", "explicit_maxThroughput", "explicit_requestsPerSecond", "provider_catalog", "generic_fallback" ], "type": "string" }, "controlPlaneFeePerHour": { "description": "Cost-only cluster management fee in USD/hr; it does not represent control-plane CPU, throttling, or recovery telemetry", "type": "number" }, "currentNodesCostPerHour": { "description": "Total cluster cost at the current node count (control plane + nodes × rate) × spotFactor in USD/hr", "type": "number" }, "effectiveCpuCapacityRps": { "description": "CPU-curve denominator after applying the provider threshold to explicit capacityRps", "type": "number" }, "id": { "description": "Resource ID", "type": "string" }, "maxNodes": { "description": "Autoscale ceiling (scale-out stops here)", "type": "number" }, "minNodes": { "description": "Autoscale floor (scale-in stops here)", "type": "number" }, "name": { "description": "Resource display name", "type": "string" }, "nodeCount": { "description": "Current node count", "type": "number" }, "nodePools": { "description": "Per-pool billing details for multi-pool clusters (absent on single-pool clusters)", "items": { "additionalProperties": true, "properties": { "maxNodes": { "type": "number" }, "minNodes": { "type": "number" }, "name": { "type": "string" }, "nodeCount": { "type": "number" }, "perNodeMaxThroughputRps": { "type": "number" }, "perNodeRatePerHour": { "type": "number" } }, "required": [ "nodeCount", "minNodes", "maxNodes", "perNodeRatePerHour" ], "type": "object" }, "type": "array" }, "perNodeRatePerHour": { "description": "Per-node billing rate in USD/hr", "type": "number" }, "perNodeTokensPerSec": { "description": "Modelled per-node token throughput at full utilisation (tokens/sec) — present only on inference-mode kubernetes resources", "type": "number" }, "provider": { "description": "Cloud provider", "type": "string" }, "rateProvenance": { "additionalProperties": false, "description": "Trust metadata for the resolved rate: pricing basis (on-demand/spot/estimated), lookup path, and the date the constant was last verified against the provider's public pricing page. Present on compute, database, and GPU inference resources; absent on resource types where no rate is resolved.", "properties": { "architecture": { "description": "Fargate task architecture when architecture-specific billing applies", "type": "string" }, "basis": { "description": "Pricing basis: published on-demand list price, spot, or a flat estimate", "enum": [ "on-demand", "spot", "estimated" ], "type": "string" }, "lastVerified": { "description": "ISO-8601 date the constant was last cross-checked against the provider's pricing page; null when unknown", "type": [ "string", "null" ] }, "region": { "description": "Region the verified rate applies to; null for region-uniform pricing", "type": [ "string", "null" ] }, "resolution": { "description": "Which lookup path resolved the rate", "enum": [ "large-sku", "entry-level-fallback", "base-rate-multiplier", "fargate-task-vcpu-memory", "aurora-acu-hour", "cloud-run-request-based", "request-serving-usage-based" ], "type": "string" }, "sku": { "description": "Exact catalog SKU used for this rate, when available", "type": "string" }, "source": { "description": "Official pricing source URL, when verified", "type": "string" } }, "required": [ "basis", "resolution", "lastVerified", "region" ], "type": "object" }, "requestServing": { "additionalProperties": true, "description": "Resolved request-serving settings. For Fargate, aggregateCapacityRps is desiredTaskCount × this service's own rate, maxAggregateCapacityRps uses maxTaskCount, and perTaskCapacityEvidence distinguishes assumptions from caller-asserted documentation/measurement or provider catalog data.", "properties": { "aggregateCapacityRps": { "minimum": 0, "type": "number" }, "desiredTaskCount": { "minimum": 0, "type": "integer" }, "maxAggregateCapacityRps": { "minimum": 0, "type": "number" }, "maxTaskCount": { "exclusiveMinimum": 0, "type": "integer" }, "minTaskCount": { "minimum": 0, "type": "integer" }, "perTaskCapacityEvidence": { "additionalProperties": false, "properties": { "basis": { "enum": [ "assumption", "documented", "measured", "catalog" ], "type": "string" }, "origin": { "enum": [ "caller-provided", "fargate-size-heuristic", "policy-default", "provider-catalog" ], "type": "string" }, "source": { "type": "string" } }, "required": [ "basis", "origin" ], "type": "object" }, "perTaskCapacityRps": { "exclusiveMinimum": 0, "type": "number" }, "serviceFamily": { "type": "string" } }, "type": "object" }, "resolvedConnectionLimit": { "description": "Effective max-connection limit the engine uses for connection-pressure modelling (database resources)", "type": "number" }, "resolvedCostMultiplier": { "description": "Effective cost multiplier the engine applies to the base provider rate for this resource", "type": "number" }, "resolvedGpuRatePerNode": { "description": "Resolved GPU node billing rate in USD/hr — present only on inference-mode kubernetes resources", "type": "number" }, "resolvedHourlyRate": { "description": "Resolved hourly billing rate in USD/hr (base rate × multiplier)", "type": "number" }, "resolvedMaxThroughputRps": { "description": "Per-node RPS ceiling the engine uses for load and autoscale calculations", "type": "number" }, "resolvedSkuLabel": { "description": "The GPU SKU that was matched (e.g. 'a100-80gb') or a fallback label naming the provider default — present only on inference-mode kubernetes resources", "type": "string" }, "resolvedStorageTier": { "description": "Storage tier / size string used to resolve pricing (database resources)", "type": "string" }, "spotFactor": { "description": "Spot-instance discount factor applied to the node cost (absent = 1, i.e. no discount)", "type": "number" }, "type": { "description": "Resource type (compute, database, kubernetes, …)", "type": "string" } }, "type": "object" }, "type": "array" } }, "type": "object" }, "predictionEffectiveConfigHash": { "description": "Versioned prediction hash over replay startup inputs, engine version, and calibration identity.", "maxLength": 64, "minLength": 64, "type": "string" }, "predictionEvidence": { "anyOf": [ { "additionalProperties": false, "properties": { "appCpu": { "anyOf": [ { "additionalProperties": false, "properties": { "central": { "minimum": 0, "type": "number" }, "high": { "minimum": 0, "type": "number" }, "low": { "minimum": 0, "type": "number" } }, "required": [ "low", "central", "high" ], "type": "object" }, { "type": "null" } ] }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "legacyGeneric": { "type": "boolean" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId", "legacyGeneric" ], "type": "object" }, "type": "array" }, "appWeight": { "enum": [ "lean", "typical", "heavy" ], "type": "string" }, "appWeightDefaulted": { "type": "boolean" }, "evidenceLevel": { "enum": [ "measured", "scaled from measured", "reference estimate" ], "type": "string" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyAvailability": { "additionalProperties": false, "properties": { "reason": { "minLength": 1, "type": "string" }, "resourceIds": { "items": { "minLength": 1, "type": "string" }, "type": "array" }, "status": { "const": "unavailable", "type": "string" } }, "required": [ "status", "reason", "resourceIds" ], "type": "object" }, "latencyP50": { "anyOf": [ { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" } }, "required": [ "low", "central", "high", "evidenceLevel" ], "type": "object" }, { "type": "null" } ] }, "latencyP95": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP99": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyPercentileStatus": { "enum": [ "modeled", "unavailable" ], "type": "string" }, "legacyGeneric": { "type": "boolean" }, "loadScope": { "enum": [ "within measured load", "beyond measured load", "below measured load", "not calibrated" ], "type": "string" }, "note": { "type": "string" }, "requestServingCapacity": { "items": { "additionalProperties": false, "properties": { "aggregateCapacityRps": { "minimum": 0, "type": "number" }, "capacityEvidence": { "additionalProperties": false, "properties": { "basis": { "enum": [ "assumption", "documented", "measured", "catalog" ], "type": "string" }, "origin": { "enum": [ "caller-provided", "fargate-size-heuristic", "policy-default", "provider-catalog" ], "type": "string" }, "source": { "type": "string" } }, "required": [ "basis", "origin" ], "type": "object" }, "maxAggregateCapacityRps": { "minimum": 0, "type": "number" }, "maxTaskCount": { "exclusiveMinimum": 0, "type": "integer" }, "perTaskCapacityRps": { "exclusiveMinimum": 0, "type": "number" }, "resourceId": { "type": "string" }, "resourceName": { "type": "string" }, "taskCount": { "minimum": 0, "type": "integer" } }, "required": [ "resourceId", "resourceName", "taskCount", "perTaskCapacityRps", "aggregateCapacityRps", "maxTaskCount", "maxAggregateCapacityRps", "capacityEvidence" ], "type": "object" }, "type": "array" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "legacyGeneric", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" }, { "additionalProperties": false, "properties": { "appCpu": { "type": "null" }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId" ], "type": "object" }, "maxItems": 0, "type": "array" }, "appWeight": { "type": "null" }, "appWeightDefaulted": { "type": "null" }, "evidenceLevel": { "const": "unavailable/legacy", "type": "string" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyP50": { "type": "null" }, "latencyP95": { "type": "null" }, "latencyP99": { "type": "null" }, "legacyGeneric": { "type": "null" }, "loadScope": { "type": "null" }, "note": { "type": "string" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" }, { "additionalProperties": false, "properties": { "appCpu": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0" }, { "type": "null" } ] }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "legacyGeneric": { "type": "boolean" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId" ], "type": "object" }, "type": "array" }, "appWeight": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appWeight" }, "appWeightDefaulted": { "type": "boolean" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyAvailability": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyAvailability" }, "latencyP50": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP95": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP99": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyPercentileStatus": { "enum": [ "modeled", "unavailable" ], "type": "string" }, "legacyGeneric": { "type": "boolean" }, "loadScope": { "enum": [ "within measured load", "beyond measured load", "below measured load", "not calibrated" ], "type": "string" }, "note": { "type": "string" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" } ] }, "replayIdentity": { "additionalProperties": false, "properties": { "effectiveConfigHash": { "description": "Original replay-only SHA-256 of the effective six-control startup configuration; top-level effectiveConfigHash is the versioned prediction identity.", "maxLength": 64, "minLength": 64, "type": "string" }, "scenarioHash": { "description": "Canonical SHA-256 of the persisted scenario graph and attached traffic-pattern order", "maxLength": 64, "minLength": 64, "type": "string" } }, "required": [ "scenarioHash", "effectiveConfigHash" ], "type": "object" }, "resilienceConfig": { "additionalProperties": false, "description": "Effective resilience model returned after create; per-edge retryPolicy values include defaults for omitted fields.", "properties": { "dependencies": { "description": "Dependency edges with retry and protection policies", "items": { "additionalProperties": false, "properties": { "authDependencyId": { "description": "Dependency receiving generated auth/token traffic", "maxLength": 128, "minLength": 1, "type": "string" }, "authRequestsPerAttempt": { "description": "Auth/token requests generated per dependency attempt", "maximum": 10, "minimum": 0, "type": "number" }, "capacity": { "additionalProperties": false, "properties": { "maxConcurrent": { "exclusiveMinimum": 0, "maximum": 1000000, "type": "integer" }, "maxRps": { "exclusiveMinimum": 0, "maximum": 500000, "type": "number" }, "meanServiceTimeMs": { "exclusiveMinimum": 0, "maximum": 120000, "type": "number" } }, "type": "object" }, "id": { "description": "Unique dependency edge ID", "maxLength": 128, "minLength": 1, "type": "string" }, "protection": { "additionalProperties": false, "properties": { "circuitBreaker": { "additionalProperties": false, "properties": { "enabled": { "type": "boolean" }, "failureRateThreshold": { "maximum": 1, "minimum": 0, "type": "number" }, "halfOpenMaxRequests": { "maximum": 10000, "minimum": 1, "type": "integer" }, "minimumRequests": { "maximum": 100000, "minimum": 1, "type": "integer" }, "openSteps": { "maximum": 120, "minimum": 1, "type": "integer" } }, "type": "object" }, "loadShedding": { "type": "boolean" }, "rateLimitRps": { "exclusiveMinimum": 0, "maximum": 500000, "type": "number" } }, "type": "object" }, "requestRatio": { "description": "Requests to target per original request", "maximum": 20, "minimum": 0, "type": "number" }, "retryPolicy": { "additionalProperties": false, "properties": { "backoffMs": { "description": "Initial retry backoff in milliseconds (default 100).", "maximum": 60000, "minimum": 0, "type": "integer" }, "backoffMultiplier": { "description": "Exponential backoff multiplier (default 2).", "maximum": 10, "minimum": 1, "type": "number" }, "jitterRatio": { "description": "Fractional backoff jitter from 0 to 1 (default 0.1).", "maximum": 1, "minimum": 0, "type": "number" }, "maxRetries": { "description": "Maximum retries for this dependency edge (0 disables retries; default 2).", "maximum": 8, "minimum": 0, "type": "integer" }, "retryActorId": { "description": "Actor producing retries, such as a gateway or client", "maxLength": 128, "minLength": 1, "type": "string" }, "retryBudgetRatio": { "description": "Retry RPS budget as a multiple of original edge RPS (default 2).", "maximum": 10, "minimum": 0, "type": "number" }, "retryBudgetRps": { "description": "Optional absolute retry RPS cap, also bounded by retryBudgetRatio.", "maximum": 500000, "minimum": 0, "type": "number" }, "timeoutMs": { "description": "Per-attempt client deadline in milliseconds (default 2000); includes queue wait, service, and latency.", "maximum": 120000, "minimum": 1, "type": "integer" } }, "type": "object" }, "sourceId": { "description": "Upstream resource ID", "minLength": 1, "type": "string" }, "targetId": { "description": "Downstream resource ID", "minLength": 1, "type": "string" } }, "required": [ "id", "sourceId", "targetId" ], "type": "object" }, "maxItems": 64, "type": "array" }, "enabled": { "description": "Master switch for retry/cascade modeling", "type": "boolean" }, "maxCascadeDepth": { "maximum": 8, "minimum": 1, "type": "integer" }, "maxGeneratedRps": { "exclusiveMinimum": 0, "maximum": 500000, "type": "number" }, "maxStepWork": { "maximum": 2048, "minimum": 1, "type": "integer" }, "retryGeneratedTrafficAffectsCost": { "type": "boolean" }, "scalingPolicies": { "description": "Capacity-observation policies, including intentional autoscaling blind spots", "items": { "additionalProperties": false, "properties": { "constrainedMetric": { "enum": [ "rps", "concurrency" ], "type": "string" }, "dependencyId": { "maxLength": 128, "minLength": 1, "type": "string" }, "id": { "maxLength": 128, "minLength": 1, "type": "string" }, "observationMetric": { "enum": [ "source_cpu", "capacity_utilization" ], "type": "string" }, "observedResourceId": { "minLength": 1, "type": "string" }, "scaleOutCapacityMultiplier": { "maximum": 20, "minimum": 1, "type": "number" }, "scaleOutThresholdPercent": { "maximum": 100, "minimum": 1, "type": "number" } }, "required": [ "id", "dependencyId", "constrainedMetric", "observationMetric", "scaleOutThresholdPercent" ], "type": "object" }, "maxItems": 32, "type": "array" }, "scheduledFaults": { "description": "Scheduled capacity, error, latency, concurrency, or traffic-surge faults", "items": { "additionalProperties": false, "properties": { "addedLatencyMs": { "maximum": 120000, "minimum": 0, "type": "number" }, "capacityPercent": { "maximum": 100, "minimum": 0, "type": "number" }, "dependencyId": { "type": "string" }, "endStep": { "minimum": 1, "type": "integer" }, "errorRate": { "maximum": 1, "minimum": 0, "type": "number" }, "id": { "maxLength": 128, "minLength": 1, "type": "string" }, "maxConcurrent": { "exclusiveMinimum": 0, "maximum": 1000000, "type": "integer" }, "startStep": { "minimum": 0, "type": "integer" }, "targetResourceId": { "type": "string" }, "trafficMultiplier": { "description": "traffic_surge demand multiplier", "exclusiveMinimum": 1, "maximum": 20, "type": "number" }, "type": { "description": "Fault type; traffic_surge increases root client demand", "enum": [ "capacity_limit", "concurrency_limit", "latency", "error_rate", "traffic_surge" ], "type": "string" } }, "required": [ "id", "type", "startStep" ], "type": "object" }, "maxItems": 32, "type": "array" }, "version": { "const": 1, "description": "Resilience model version", "type": "number" } }, "type": "object" }, "resources": { "description": "Per-resource summary (compact mode) or full resource states (full mode)", "items": { "additionalProperties": true, "properties": { "availabilityState": { "description": "Availability derived from routed traffic; degraded can still serve, unavailable cannot", "enum": [ "available", "degraded", "unavailable" ], "type": "string" }, "cpuPercent": { "description": "CPU utilization (%)", "type": "number" }, "id": { "description": "Resource ID", "type": "string" }, "isRoutable": { "description": "Whether this compute/Kubernetes resource can receive traffic", "type": "boolean" }, "name": { "description": "Resource display name", "type": "string" }, "recoveryBlockedReason": { "description": "Engine recovery guard currently blocking cooldown progress, when present", "type": "string" }, "routedRps": { "description": "Requests per second routed to this resource (compute/kubernetes only)", "type": "number" }, "status": { "description": "Health status (healthy/warning/critical/failed)", "type": "string" } }, "type": "object" }, "type": "array" }, "scenarioAttribution": { "additionalProperties": false, "description": "Trusted server-side attribution copied from the live scenario catalog; absent for explicit resource-graph creates", "properties": { "id": { "description": "Live scenario catalog ID that supplied the simulation graph", "type": "string" }, "revision": { "description": "Catalog revision, when provided", "type": [ "string", "null" ] }, "source": { "const": "scenario-catalog", "description": "Trusted server-side attribution source", "type": "string" }, "version": { "description": "Catalog version, when provided", "type": [ "string", "null" ] } }, "required": [ "id", "version", "revision", "source" ], "type": "object" }, "scenarioHash": { "description": "Canonical SHA-256 of the persisted scenario graph and attached traffic-pattern order", "maxLength": 64, "minLength": 64, "type": "string" }, "status": { "description": "Current simulation status", "type": "string" }, "traffic": { "description": "Current traffic in RPS", "type": "number" } }, "type": "object" } }, { "description": "Permanently delete an owned temporary anonymous demo simulation and its metrics, events, failures, and capability. This is the explicit way to free a simulation slot; deletion is irreversible, while the existing demo TTL remains the safety net for abandoned simulations. Prerequisite: a simulationId from simulation.create, or an active simulation in the preserved MCP session. The likely next tool is simulation.create to use the freed slot. A successful response is { deleted: true, id }; failed ownership checks do not delete or revoke anything. Authenticate with an API key to unlock all 63 tools and persistent simulation management.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "simulationId": { "description": "Simulation ID returned by simulation.create. Preserve Mcp-Session-Id to omit this field and use the session's current simulation; if your connector starts a fresh MCP session for each call (for example Grok Bot or Cursor), pass this explicit ID after every fresh initialization. A fresh session has no current-simulation pointer and returns NO_ACTIVE_SIMULATION when the ID is omitted. Anonymous capabilities are short-lived (30 minutes by default), unguessable, and revoked when the demo expires or is deleted; proxy IP changes do not invalidate them. Do not treat the ID as a durable share link.", "type": "string" } }, "type": "object" }, "name": "simulation.delete", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "properties": { "deleted": { "description": "True when the simulation was deleted", "type": "boolean" }, "id": { "description": "ID of the deleted simulation", "type": "string" }, "simulationId": { "description": "ID used for the deletion", "type": "string" }, "simulationIdSource": { "enum": [ "explicit", "session_default" ], "type": "string" } }, "type": "object" } }, { "description": "Fail one compute/Kubernetes node or database in a temporary anonymous demo simulation (marked critical, not removed). Targeted database quick failures are bounded and reversible; with no serving database at positive load, simulation.step reports 100% errors, zero goodput and errorBreakdown.dbFailure: 100. simulation.recover_resource can restore the database early. An AWS Aurora writer with characteristics.auroraStandbyResourceId pointing to a healthy related replicaOf database has an opt-in modeled failover: the first step is unavailable, the second shows the standby serving without a residual writer-outage penalty. MultiAz or an unrelated second database alone does not establish a standby. Read metrics.databases[].auroraFailover and the promotion event for the failed and standby IDs and success, plus errorRate and throughput. Each quick step is one modeled simulation second, so second-step promotion is one second after injection. Chaos database_crash samples every 10 seconds and defaults to a 30-second promotion and a 1800-second sole-writer restart after its injection duration; compare matching phases, not equal step indices or wall-clock times. These are deterministic assumptions, not observed AWS behavior. No API key required for this temporary anonymous demo operation. Exact targeting: pass resourceName (human-readable name, e.g. 'app-server-01'; exact match preferred, an unambiguous prefix is accepted) or resourceId to fail a specific resource — including an individual named instance, not only a group. If resourceName matches multiple resources the call fails with a 400 listing every matching candidate by name — retry with one exact name (or its resourceId) from that list. The database must be healthy; an already failed database returns a 400 describing its current status. When neither parameter is supplied, a RANDOM healthy compute/Kubernetes node is selected (not a database) — this path is non-deterministic and NOT suitable for controlled scenarios or replay; always target by name/id when reproducing a precise fault sequence. The response always echoes the applied outcome via resolvedResourceId, resolvedResourceName, and previousHealth (populated from the selected resource on the random path too). Network, storage, cache, queue, and security resource types are not supported by quick injection. For typed database_overload use authenticated failure.create (unavailable anonymously); instance_kill PERMANENTLY removes the instance — failure.delete does not restore it; use instance_down instead for a reversible single-node outage. Returns the updated resource list and the failure event that was logged. The likely next tool is simulation.step to observe how the architecture degrades under failure, then simulation.metrics to review the health impact. Do not use it to advance simulation time — that is simulation.step. Pass simulationId from simulation.create when this call is made from a fresh MCP session; otherwise you may omit it to target the current simulation in the preserved MCP session. Authenticate with an API key to unlock all 63 tools including typed durational failures and chaos engineering.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "resourceId": { "description": "Optional: ID of the resource to fail. Takes precedence over resourceName.", "type": "string" }, "resourceName": { "description": "Optional: name of the resource to fail (exact match preferred; unambiguous prefix accepted). Ambiguous names return a 400 with a candidate list.", "type": "string" }, "simulationId": { "description": "Simulation ID returned by simulation.create. Preserve Mcp-Session-Id to omit this field and use the session's current simulation; if your connector starts a fresh MCP session for each call (for example Grok Bot or Cursor), pass this explicit ID after every fresh initialization. A fresh session has no current-simulation pointer and returns NO_ACTIVE_SIMULATION when the ID is omitted. Anonymous capabilities are short-lived (30 minutes by default), unguessable, and revoked when the demo expires or is deleted; proxy IP changes do not invalidate them. Do not treat the ID as a durable share link.", "type": "string" } }, "type": "object" }, "name": "simulation.inject_failure", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "properties": { "event": { "additionalProperties": {}, "description": "Failure event that was logged", "type": "object" }, "previousHealth": { "description": "The resource's health status immediately before the failure was applied", "type": "string" }, "resolvedResourceId": { "description": "ID of the resource that was failed (targeted or randomly selected)", "type": "string" }, "resolvedResourceName": { "description": "Name of the resource that was failed", "type": "string" }, "resources": { "description": "Updated resource list after failure injection", "items": { "additionalProperties": {}, "type": "object" }, "type": "array" } }, "type": "object" } }, { "description": "Change the traffic load on a demo simulation. Omit traffic to trigger a random 2×–5× spike (sends random: true internally); provide traffic to set an absolute RPS level (capped at 10000 RPS in demo mode). Use it to stress-test the architecture before stepping; the change only affects metrics after the next simulation.step. Do not use it to read metrics (simulation.metrics) or advance time (simulation.step). Pass the simulationId returned by simulation.create when your connector opens a fresh MCP session; preserve Mcp-Session-Id to use the omitted-ID current-simulation default. Returns the updated simulation with its new traffic level; the likely next tool is simulation.step.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "simulationId": { "description": "Simulation ID returned by simulation.create. Preserve Mcp-Session-Id to omit this field and use the session's current simulation; if your connector starts a fresh MCP session for each call (for example Grok Bot or Cursor), pass this explicit ID after every fresh initialization. A fresh session has no current-simulation pointer and returns NO_ACTIVE_SIMULATION when the ID is omitted. Anonymous capabilities are short-lived (30 minutes by default), unguessable, and revoked when the demo expires or is deleted; proxy IP changes do not invalidate them. Do not treat the ID as a durable share link.", "type": "string" }, "traffic": { "description": "Absolute traffic level in RPS to set. Omit to trigger a random spike instead. Server-capped at 10000 RPS in demo mode.", "type": "number" } }, "type": "object" }, "name": "simulation.inject_traffic", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "properties": { "id": { "description": "Simulation ID", "type": "string" }, "status": { "description": "Updated simulation status", "type": "string" }, "traffic": { "description": "New traffic level in RPS after injection", "type": "number" } }, "type": "object" } }, { "description": "Read the latest metrics and resource states for a temporary anonymous demo simulation: latency, CPU, throughput, error rate, cost per hour, and per-resource health. Use it to inspect current state and metrics history without advancing time; do not use it to move the simulation forward — that is simulation.step. Responses are compact by default: principal current metrics plus explicit modeled goodputRps (a post-step point rate sourced from throughput, with provenance), goodputWindow (recorded only from persisted simulation-clock bounds, otherwise unavailable with provenance; never derive it from retrieval time or currentStep), errorBreakdown when available, per-resource status (id, name, status, cpuPercent, routedRps, availabilityState, isRoutable, and recoveryBlockedReason when provided), seeded EKS Spot checkpoint history and migrationEvaluationComplete/provenance when present, and the last 10 metrics-history entries. Pass responseMode: 'full' to get the complete simulation state, normalizedConfig, and full metrics history instead. During recovery, each resource may include recoveryProgress.state (parked, cooling_down, or healthy) with parkWindow and cooldown counters; poll simulation.step until healthy, then use simulation.metrics to inspect the resulting state and metrics. GPU / inference workflow: when the simulation includes a kubernetes resource with characteristics.inferenceMode: true, the response also includes top-level gpuUtilization (%), tokensPerSecond, costPerMillionTokens (USD/M tokens), idleGpuCostPerHour (USD/hr of standby GPU spend), and idleGpuFraction (0-1 idle HA overhead share) from the latest step, and each history entry carries the same inference fields. Pass the simulationId returned by simulation.create when your connector opens a fresh MCP session; preserve Mcp-Session-Id to use the omitted-ID current-simulation default. A fresh session has no current-simulation pointer. At least one simulation.step is needed for meaningful metrics. Read-only; repeated calls may be subject to demo usage limits. The likely next tool is simulation.step or simulation.inject_traffic.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "responseMode": { "default": "compact", "description": "Response detail level. 'compact' (default) returns principal current metrics, errorBreakdown when available, per-resource status (id, name, status, cpuPercent, routedRps, availabilityState, isRoutable, recoveryBlockedReason, and failureLifecycle/routingState when provided), and only the last 10 metrics-history entries — keeps polling cheap for agent loops. 'full' returns the complete simulation object (all resource characteristics and connections) plus the entire metrics history.", "enum": [ "compact", "full" ], "type": "string" }, "simulationId": { "description": "Simulation ID returned by simulation.create. Preserve Mcp-Session-Id to omit this field and use the session's current simulation; if your connector starts a fresh MCP session for each call (for example Grok Bot or Cursor), pass this explicit ID after every fresh initialization. A fresh session has no current-simulation pointer and returns NO_ACTIVE_SIMULATION when the ID is omitted. Anonymous capabilities are short-lived (30 minutes by default), unguessable, and revoked when the demo expires or is deleted; proxy IP changes do not invalidate them. Do not treat the ID as a durable share link.", "type": "string" } }, "type": "object" }, "name": "simulation.metrics", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "description": "Compact metrics summary by default (principal current metrics, per-resource status, bounded history tail). With responseMode 'full', the complete simulation object and full metrics history are passed through instead.", "properties": { "appWeight": { "enum": [ "lean", "typical", "heavy" ], "type": "string" }, "appWeightDefaulted": { "type": "boolean" }, "auroraFailovers": { "description": "Latest compact per-writer Aurora failover state, with failed and standby IDs and phase.", "items": { "additionalProperties": false, "properties": { "failedResourceId": { "type": "string" }, "phase": { "enum": [ "promoting", "serving", "unavailable" ], "type": "string" }, "standbyResourceId": { "type": "string" }, "succeeded": { "type": "boolean" } }, "required": [ "failedResourceId", "standbyResourceId", "succeeded", "phase" ], "type": "object" }, "type": "array" }, "calibrationEvidence": { "additionalProperties": false, "description": "Owned-versus-modeled evidence and latency boundary.", "properties": { "calibrationId": { "description": "Versioned owned calibration identifier; present only when that calibration applies.", "type": "string" }, "kind": { "description": "owned for the exact AWS CRUD fit; modeled when the gate fails or another generic model applies.", "enum": [ "owned", "modeled" ], "type": "string" }, "latencyBasis": { "description": "General modeled latency path, such as in-VPC ALB rather than end-to-end; see latencyP99Basis for percentile-specific P99 provenance.", "type": "string" }, "latencyP99Basis": { "description": "Percentile-specific P99 basis: owned in-VPC internal-ALB fit, scaled from that fit (not directly measured), or uncalibrated generic model.", "type": "string" }, "note": { "description": "Names the fit scope and limitations. For canonical workload inference, states it is a modeling assumption. For modeled fallback, names all actual failed checks in stable order with resource IDs and safe expected/actual values; never only a generic 'not applied'.", "type": "string" } }, "required": [ "kind", "latencyBasis", "note" ], "type": "object" }, "costPerHour": { "description": "Latest estimated cost in USD/hr", "type": "number" }, "costPerMillionTokens": { "description": "Latest self-hosted inference cost in USD per million tokens (null when no tokens are being processed) — present only on GPU inference simulations", "type": [ "number", "null" ] }, "currentStep": { "description": "Current simulation time step", "type": "number" }, "effectiveConfigHash": { "description": "Versioned prediction hash over replay startup inputs, engine version, and calibration identity.", "maxLength": 64, "minLength": 64, "type": "string" }, "eksSpotInterruptions": { "description": "Latest seeded EKS interruption telemetry, including additive migrationEvaluation/provenance when recorded.", "items": { "additionalProperties": true, "properties": { "checkpointEvidence": { "additionalProperties": false, "description": "Persisted seeded-EKS checkpoint captured with its interruption JSONB record. engineInputStepIndex is an engine input-step index, simulationSeconds is derived from that index and the recorded tick, and traffic provenance points to outer metric fields rather than duplicating float values.", "properties": { "engineInputStepIndex": { "minimum": 0, "type": "integer" }, "fieldProvenance": { "additionalProperties": false, "properties": { "engineInputStepIndex": { "$ref": "#/properties/goodputProvenance" }, "intervalAttribution": { "$ref": "#/properties/goodputProvenance" }, "pointRateSemantics": { "$ref": "#/properties/goodputProvenance" }, "simulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "tickDurationSeconds": { "$ref": "#/properties/goodputProvenance" } }, "required": [ "engineInputStepIndex", "simulationSeconds", "tickDurationSeconds", "pointRateSemantics", "intervalAttribution" ], "type": "object" }, "intervalAttribution": { "anyOf": [ { "anyOf": [ { "additionalProperties": false, "properties": { "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" }, "source": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "source" ], "type": "object" }, "status": { "const": "recorded", "type": "string" }, "window": { "$ref": "#/properties/goodputWindow/anyOf/1/properties/window" } }, "required": [ "status", "window", "provenance" ], "type": "object" }, { "additionalProperties": false, "properties": { "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" }, "status": { "const": "unavailable", "type": "string" } }, "required": [ "status", "provenance" ], "type": "object" } ] }, { "type": "null" } ] }, "pointRateSemantics": { "const": "post_step_point_rate", "type": "string" }, "simulationSeconds": { "minimum": 0, "type": "number" }, "tickDurationSeconds": { "exclusiveMinimum": 0, "type": "number" }, "trafficFieldProvenance": { "additionalProperties": false, "properties": { "errorRatePercent": { "$ref": "#/properties/goodputProvenance" }, "latencyP50Ms": { "$ref": "#/properties/goodputProvenance" }, "offeredRps": { "$ref": "#/properties/goodputProvenance" }, "throughputRps": { "$ref": "#/properties/goodputProvenance" } }, "required": [ "throughputRps", "offeredRps", "errorRatePercent", "latencyP50Ms" ], "type": "object" } }, "required": [ "engineInputStepIndex", "simulationSeconds", "tickDurationSeconds", "pointRateSemantics", "intervalAttribution", "fieldProvenance", "trafficFieldProvenance" ], "type": "object" }, "migrationEvaluation": { "additionalProperties": false, "description": "Authoritative additive seeded EKS interruption evaluation. Field values carry recorded/derived/unavailable provenance; deadline verdict/counts/reasons are frozen once status is completed and do not describe later service health.", "properties": { "affectedWorkloadCount": { "anyOf": [ { "minimum": 0, "type": "integer" }, { "type": "null" } ] }, "allAffectedWorkloadsReadyAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "deadlineAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "deadlineMet": { "type": [ "boolean", "null" ] }, "deadlineSeconds": { "const": 120, "type": "number" }, "evaluatedAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "fieldProvenance": { "additionalProperties": false, "properties": { "affectedWorkloadCount": { "$ref": "#/properties/goodputProvenance" }, "allAffectedWorkloadsReadyAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "deadlineAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "deadlineMet": { "$ref": "#/properties/goodputProvenance" }, "deadlineSeconds": { "$ref": "#/properties/goodputProvenance" }, "evaluatedAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "interruptionNoticeAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "limitingReasons": { "$ref": "#/properties/goodputProvenance" }, "migrationDurationSeconds": { "$ref": "#/properties/goodputProvenance" }, "migrationStartedAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "missedDeadlineWorkloadCount": { "$ref": "#/properties/goodputProvenance" }, "readyWorkloadCountAtDeadline": { "$ref": "#/properties/goodputProvenance" }, "status": { "$ref": "#/properties/goodputProvenance" } }, "required": [ "status", "interruptionNoticeAtSimulationSeconds", "deadlineAtSimulationSeconds", "migrationStartedAtSimulationSeconds", "allAffectedWorkloadsReadyAtSimulationSeconds", "migrationDurationSeconds", "deadlineSeconds", "deadlineMet", "affectedWorkloadCount", "readyWorkloadCountAtDeadline", "missedDeadlineWorkloadCount", "evaluatedAtSimulationSeconds", "limitingReasons" ], "type": "object" }, "interruptionNoticeAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "limitingReasons": { "items": { "additionalProperties": false, "properties": { "code": { "enum": [ "no_reschedule", "scheduling_capacity_exhausted", "image_pull_incomplete", "startup_incomplete", "unclassified_runtime_work_remaining" ], "type": "string" }, "interruptionHandling": { "enum": [ "reschedule", "drain-only", "fail-fast" ], "type": "string" }, "pendingWorkloadCount": { "minimum": 0, "type": "integer" }, "provenance": { "const": "recorded", "type": "string" }, "pullingWorkloadCount": { "minimum": 0, "type": "integer" }, "residualPullSeconds": { "minimum": 0, "type": "number" }, "residualStartupSeconds": { "minimum": 0, "type": "number" }, "schedulingCapacity": { "minimum": 0, "type": "integer" }, "startingWorkloadCount": { "minimum": 0, "type": "integer" }, "workloadCount": { "minimum": 0, "type": "integer" } }, "required": [ "code", "workloadCount", "pendingWorkloadCount", "pullingWorkloadCount", "startingWorkloadCount", "residualPullSeconds", "residualStartupSeconds", "schedulingCapacity", "interruptionHandling", "provenance" ], "type": "object" }, "maxItems": 5, "type": "array" }, "migrationDurationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "migrationStartedAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "missedDeadlineWorkloadCount": { "anyOf": [ { "minimum": 0, "type": "integer" }, { "type": "null" } ] }, "readyWorkloadCountAtDeadline": { "anyOf": [ { "minimum": 0, "type": "integer" }, { "type": "null" } ] }, "status": { "enum": [ "not_started", "in_progress", "completed" ], "type": "string" } }, "required": [ "status", "interruptionNoticeAtSimulationSeconds", "deadlineAtSimulationSeconds", "migrationStartedAtSimulationSeconds", "allAffectedWorkloadsReadyAtSimulationSeconds", "migrationDurationSeconds", "deadlineSeconds", "deadlineMet", "affectedWorkloadCount", "readyWorkloadCountAtDeadline", "missedDeadlineWorkloadCount", "evaluatedAtSimulationSeconds", "limitingReasons", "fieldProvenance" ], "type": "object" }, "name": { "type": "string" }, "resourceId": { "type": "string" } }, "type": "object" }, "type": "array" }, "engineVersion": { "description": "Simulation engine version used for this prediction.", "type": "string" }, "errorBreakdown": { "additionalProperties": false, "description": "Validated additive error contributors from the latest metrics entry, in percentage-point units", "properties": { "capacityOverload": { "type": "number" }, "computeFailure": { "type": "number" }, "cpuOverload": { "type": "number" }, "dbFailure": { "type": "number" }, "dependencyFailure": { "type": "number" }, "ociStorage": { "type": "number" }, "poolSaturation": { "type": "number" }, "queueAbsorption": { "type": "number" }, "runtimeMemory": { "type": "number" } }, "required": [ "poolSaturation", "dbFailure", "computeFailure", "capacityOverload", "cpuOverload", "ociStorage", "queueAbsorption" ], "type": "object" }, "errorRate": { "description": "Latest client-level error rate (%) against original offered RPS. Resilience dependency targets in autoscaled compute fleets use aggregate routable-fleet capacity and health; non-autoscaled targets use their resolved resource capacity, including any declared fixed-instance count.", "type": "number" }, "goodputProvenance": { "anyOf": [ { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" } }, "required": [ "kind" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "derived", "type": "string" }, "sourceFields": { "items": { "minLength": 1, "type": "string" }, "minItems": 1, "type": "array" } }, "required": [ "kind", "sourceFields" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" } ], "description": "Provenance for the modeled goodput field" }, "goodputRps": { "description": "Modeled client-level successful requests per second; a post-step point rate sourced from metrics.throughput. Resilience dependency targets in autoscaled compute fleets use aggregate routable-fleet capacity and health; non-autoscaled targets use their resolved resource capacity, including any declared fixed-instance count.", "type": "number" }, "goodputSemantics": { "const": "post_step_point_rate", "description": "Goodput is a point rate, not an interval total", "type": "string" }, "goodputWindow": { "anyOf": [ { "additionalProperties": false, "properties": { "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" }, "status": { "const": "unavailable", "type": "string" } }, "required": [ "status", "provenance" ], "type": "object" }, { "additionalProperties": false, "properties": { "aggregate": { "additionalProperties": false, "properties": { "coveredDurationSeconds": { "minimum": 0, "type": "number" }, "goodputRequestTotal": { "minimum": 0, "type": "number" }, "goodputRps": { "minimum": 0, "type": "number" }, "provenance": { "anyOf": [ { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" }, "source": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "source" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "derived", "type": "string" }, "sourceFields": { "items": { "maxLength": 256, "minLength": 1, "type": "string" }, "maxItems": 256, "minItems": 1, "type": "array" } }, "required": [ "kind", "sourceFields" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" } ] }, "sampleIds": { "items": { "minLength": 1, "type": "string" }, "type": "array" }, "window": { "$ref": "#/properties/goodputWindow/anyOf/1/properties/window" } }, "required": [ "window", "coveredDurationSeconds", "goodputRps", "goodputRequestTotal", "sampleIds", "provenance" ], "type": "object" }, "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" }, "source": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "source" ], "type": "object" }, "status": { "const": "recorded", "type": "string" }, "window": { "additionalProperties": false, "properties": { "inclusion": { "const": "[start,end)", "type": "string" }, "windowEndSeconds": { "minimum": 0, "type": "number" }, "windowStartSeconds": { "minimum": 0, "type": "number" } }, "required": [ "windowStartSeconds", "windowEndSeconds", "inclusion" ], "type": "object" } }, "required": [ "status", "window", "aggregate", "provenance" ], "type": "object" } ], "description": "Interval goodput aggregate when every point has persisted simulation-clock bounds; otherwise status=unavailable. Never derive this from retrieval time or currentStep." }, "gpuUtilization": { "description": "Latest GPU utilization (%) — present only on GPU inference simulations", "type": "number" }, "idleGpuCostPerHour": { "description": "Latest USD/hr of GPU spend funding idle/standby capacity (HA overhead) — present only on GPU inference simulations", "type": "number" }, "idleGpuFraction": { "description": "Latest share (0-1) of the GPU bill that is idle/standby capacity — present only on GPU inference simulations", "type": "number" }, "latencyAvailability": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyAvailability" }, "latencyBasis": { "description": "Latest general modeled latency path/boundary; see latencyP99Basis for P99-specific provenance.", "type": "string" }, "latencyP50": { "description": "Latest 50th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP95": { "description": "Latest 95th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP99": { "description": "Latest modeled 99th-percentile latency in ms; null when workload-specific latency evidence is unavailable. Interpret with latencyP99Basis and predictionEvidence.latencyP99.", "type": [ "number", "null" ] }, "latencyP99Basis": { "description": "Latest percentile-specific P99 basis: owned fit, scaled-from-fit (not directly measured), or uncalibrated generic model.", "type": "string" }, "metricId": { "description": "Storage-assigned ID of the latest persisted metric", "type": "string" }, "metrics": { "description": "Metrics history — bounded to the last 10 entries in compact mode, full history in full mode", "items": { "additionalProperties": true, "properties": { "costPerHour": { "description": "Estimated cost in USD/hr", "type": "number" }, "costPerMillionTokens": { "description": "Self-hosted inference cost in USD per million tokens for this step — present only on GPU inference simulations", "type": [ "number", "null" ] }, "cpuUsage": { "description": "CPU utilization (%)", "type": "number" }, "eksSpotInterruptions": { "items": { "$ref": "#/properties/eksSpotInterruptions/items" }, "type": "array" }, "errorBreakdown": { "additionalProperties": false, "description": "Validated additive error contributors for this history entry", "properties": { "capacityOverload": { "type": "number" }, "computeFailure": { "type": "number" }, "cpuOverload": { "type": "number" }, "dbFailure": { "type": "number" }, "dependencyFailure": { "type": "number" }, "ociStorage": { "type": "number" }, "poolSaturation": { "type": "number" }, "queueAbsorption": { "type": "number" }, "runtimeMemory": { "type": "number" } }, "required": [ "poolSaturation", "dbFailure", "computeFailure", "capacityOverload", "cpuOverload", "ociStorage", "queueAbsorption" ], "type": "object" }, "errorRate": { "description": "Client-level error rate (%) against original offered RPS. Resilience dependency targets in autoscaled compute fleets use aggregate routable-fleet capacity and health; standalone targets count once.", "type": "number" }, "gpuUtilization": { "description": "GPU utilization (%) for this step — present only on GPU inference simulations", "type": "number" }, "latencyAvailability": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyAvailability" }, "latencyBasis": { "description": "General modeled latency path/boundary; see latencyP99Basis for P99-specific provenance.", "type": "string" }, "latencyP50": { "description": "Median latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP95": { "description": "95th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP99": { "description": "Modeled 99th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP99Basis": { "description": "Percentile-specific P99 basis for this history record.", "type": "string" }, "metricId": { "description": "Storage-assigned persisted metric ID, when this history entry was stored", "type": "string" }, "retryAmplificationFactor": { "description": "Per-step (original offered RPS + generated retry RPS) / original offered RPS — present only when resilience is enabled; an amplification measure, not capacity or goodput.", "type": [ "number", "null" ] }, "throughput": { "description": "Effective RPS", "type": "number" }, "tokensPerSecond": { "description": "Inference throughput in tokens/second for this step — present only on GPU inference simulations", "type": "number" } }, "type": "object" }, "type": "array" }, "metricsHistoryLength": { "description": "Total number of metrics-history entries (compact mode returns only the last 10)", "type": "number" }, "modeledShedRps": { "description": "Latest aggregate modeled shed requests per second", "type": "number" }, "offeredRps": { "description": "Latest aggregate offered requests per second", "type": "number" }, "predictionEffectiveConfigHash": { "description": "Versioned prediction hash over replay startup inputs, engine version, and calibration identity.", "maxLength": 64, "minLength": 64, "type": "string" }, "predictionEvidence": { "anyOf": [ { "additionalProperties": false, "properties": { "appCpu": { "anyOf": [ { "additionalProperties": false, "properties": { "central": { "minimum": 0, "type": "number" }, "high": { "minimum": 0, "type": "number" }, "low": { "minimum": 0, "type": "number" } }, "required": [ "low", "central", "high" ], "type": "object" }, { "type": "null" } ] }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "legacyGeneric": { "type": "boolean" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId", "legacyGeneric" ], "type": "object" }, "type": "array" }, "appWeight": { "enum": [ "lean", "typical", "heavy" ], "type": "string" }, "appWeightDefaulted": { "type": "boolean" }, "evidenceLevel": { "enum": [ "measured", "scaled from measured", "reference estimate" ], "type": "string" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyAvailability": { "additionalProperties": false, "properties": { "reason": { "minLength": 1, "type": "string" }, "resourceIds": { "items": { "minLength": 1, "type": "string" }, "type": "array" }, "status": { "const": "unavailable", "type": "string" } }, "required": [ "status", "reason", "resourceIds" ], "type": "object" }, "latencyP50": { "anyOf": [ { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" } }, "required": [ "low", "central", "high", "evidenceLevel" ], "type": "object" }, { "type": "null" } ] }, "latencyP95": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP99": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyPercentileStatus": { "enum": [ "modeled", "unavailable" ], "type": "string" }, "legacyGeneric": { "type": "boolean" }, "loadScope": { "enum": [ "within measured load", "beyond measured load", "below measured load", "not calibrated" ], "type": "string" }, "note": { "type": "string" }, "requestServingCapacity": { "items": { "additionalProperties": false, "properties": { "aggregateCapacityRps": { "minimum": 0, "type": "number" }, "capacityEvidence": { "additionalProperties": false, "properties": { "basis": { "enum": [ "assumption", "documented", "measured", "catalog" ], "type": "string" }, "origin": { "enum": [ "caller-provided", "fargate-size-heuristic", "policy-default", "provider-catalog" ], "type": "string" }, "source": { "type": "string" } }, "required": [ "basis", "origin" ], "type": "object" }, "maxAggregateCapacityRps": { "minimum": 0, "type": "number" }, "maxTaskCount": { "exclusiveMinimum": 0, "type": "integer" }, "perTaskCapacityRps": { "exclusiveMinimum": 0, "type": "number" }, "resourceId": { "type": "string" }, "resourceName": { "type": "string" }, "taskCount": { "minimum": 0, "type": "integer" } }, "required": [ "resourceId", "resourceName", "taskCount", "perTaskCapacityRps", "aggregateCapacityRps", "maxTaskCount", "maxAggregateCapacityRps", "capacityEvidence" ], "type": "object" }, "type": "array" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "legacyGeneric", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" }, { "additionalProperties": false, "properties": { "appCpu": { "type": "null" }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId" ], "type": "object" }, "maxItems": 0, "type": "array" }, "appWeight": { "type": "null" }, "appWeightDefaulted": { "type": "null" }, "evidenceLevel": { "const": "unavailable/legacy", "type": "string" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyP50": { "type": "null" }, "latencyP95": { "type": "null" }, "latencyP99": { "type": "null" }, "legacyGeneric": { "type": "null" }, "loadScope": { "type": "null" }, "note": { "type": "string" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" }, { "additionalProperties": false, "properties": { "appCpu": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0" }, { "type": "null" } ] }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "legacyGeneric": { "type": "boolean" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId" ], "type": "object" }, "type": "array" }, "appWeight": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appWeight" }, "appWeightDefaulted": { "type": "boolean" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyAvailability": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyAvailability" }, "latencyP50": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP95": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP99": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyPercentileStatus": { "enum": [ "modeled", "unavailable" ], "type": "string" }, "legacyGeneric": { "type": "boolean" }, "loadScope": { "enum": [ "within measured load", "beyond measured load", "below measured load", "not calibrated" ], "type": "string" }, "note": { "type": "string" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" } ] }, "replayIdentity": { "additionalProperties": false, "properties": { "effectiveConfigHash": { "description": "Original replay-only SHA-256 of the effective six-control startup configuration; top-level effectiveConfigHash is the versioned prediction identity.", "maxLength": 64, "minLength": 64, "type": "string" }, "scenarioHash": { "description": "Canonical SHA-256 of the persisted scenario graph and attached traffic-pattern order", "maxLength": 64, "minLength": 64, "type": "string" } }, "required": [ "scenarioHash", "effectiveConfigHash" ], "type": "object" }, "resilienceDiagnostics": { "additionalProperties": true, "description": "Bounded resilience diagnostics from the latest step (compact mode). Absent when the resilience model did not run.", "properties": { "bounded": { "description": "True when a generated-traffic, traversal-work, or cascade-depth bound truncated model work", "type": "boolean" }, "capacityByResource": { "description": "Unique logical dependency resources in the latest step; each appears once even when several paths reference it. When values are finite, nominalCapacityRps minus warmupCapacityLossRps (members still warming) and healthCapacityLossRps (unavailable or degraded members) equals routableCapacityRps, within rounding.", "items": { "additionalProperties": false, "properties": { "healthCapacityLossRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity lost because members are unavailable or degraded; zero means no modeled member-health loss" }, "nominalCapacityRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity before warm-up and member-health reductions; null means no finite capacity ceiling is known" }, "resourceId": { "description": "Logical resource ID used by one or more resilience dependency paths", "type": "string" }, "role": { "description": "Whether this resource is a dependency source, target, or both", "enum": [ "source", "target", "source_and_target" ], "type": "string" }, "routableCapacityRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity available to this logical resource after modeled warm-up and member health; null means no finite capacity ceiling is known" }, "warmupCapacityLossRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity lost because fleet members are still warming up; zero means no modeled warm-up loss" } }, "required": [ "resourceId", "role", "routableCapacityRps" ], "type": "object" }, "type": "array" }, "incidentOutcome": { "description": "Latest incident outcome: stable | degraded | cascading | protected | recovered", "type": "string" }, "pathCount": { "description": "Number of dependency paths evaluated in the latest step", "type": "number" } }, "type": "object" }, "resources": { "description": "Per-resource status summary (compact mode)", "items": { "additionalProperties": true, "properties": { "availabilityState": { "description": "Availability derived from routed traffic: available = serving normally, degraded = critical/warning but still serving, unavailable = no traffic and not routable", "enum": [ "available", "degraded", "unavailable" ], "type": "string" }, "cpuPercent": { "description": "CPU utilization (%)", "type": "number" }, "failureLifecycle": { "description": "Read-only failure lifecycle marker when the backend identifies a quick injection or typed instance failure", "enum": [ "quick_injection_parked", "quick_injection_rejoined", "instance_down", "instance_kill" ], "type": "string" }, "id": { "description": "Resource ID", "type": "string" }, "isRoutable": { "description": "Whether this compute/Kubernetes resource can receive traffic in this step; false distinguishes a failed/parked node from a critical node still serving", "type": "boolean" }, "name": { "description": "Resource display name", "type": "string" }, "recoveryBlockedReason": { "description": "Engine recovery guard currently preventing cooldown progress, when present (for example failure_park_window or idle_cpu_floor)", "type": "string" }, "recoveryProgress": { "additionalProperties": false, "description": "Read-only recovery progress; poll simulation.step until state is healthy, then use simulation.metrics to inspect the result", "properties": { "cooldown": { "additionalProperties": false, "properties": { "completedSteps": { "type": "number" }, "remainingSteps": { "type": "number" }, "requiredSteps": { "type": "number" }, "target": { "anyOf": [ { "enum": [ "warning", "healthy" ], "type": "string" }, { "type": "null" } ] } }, "required": [ "target", "completedSteps", "requiredSteps", "remainingSteps" ], "type": "object" }, "parkWindow": { "additionalProperties": false, "properties": { "completedSteps": { "type": "number" }, "remainingSteps": { "type": "number" }, "totalSteps": { "type": "number" } }, "required": [ "totalSteps", "completedSteps", "remainingSteps" ], "type": "object" }, "state": { "enum": [ "parked", "blocked", "cooling_down", "healthy" ], "type": "string" } }, "required": [ "state", "parkWindow", "cooldown" ], "type": "object" }, "routedRps": { "description": "Requests per second routed to this resource (compute/kubernetes only)", "type": "number" }, "routingState": { "description": "Read-only current routing state for a resource with failure lifecycle telemetry", "enum": [ "unavailable", "serving" ], "type": "string" }, "status": { "description": "Health status (healthy/warning/critical/failed)", "type": "string" } }, "type": "object" }, "type": "array" }, "retryAmplificationFactor": { "description": "Latest (original offered RPS + generated retry RPS) / original offered RPS; an amplification measure, not capacity or goodput. Values > 1.0 = amplification risk. null = model ran but no traffic. Absent = resilience model disabled.", "type": [ "number", "null" ] }, "scenarioHash": { "description": "Canonical replay scenario graph hash.", "maxLength": 64, "minLength": 64, "type": "string" }, "simulation": { "additionalProperties": true, "description": "Complete simulation state (full mode only; absent in compact mode and when status is not_found/access_denied)", "properties": { "currentTime": { "description": "Current time step", "type": "number" }, "id": { "description": "Simulation ID", "type": "string" }, "name": { "description": "Simulation name", "type": "string" }, "resources": { "description": "Resource states with health and status", "items": { "additionalProperties": {}, "type": "object" }, "type": "array" }, "traffic": { "description": "Current RPS", "type": "number" } }, "type": "object" }, "simulationId": { "description": "ID of the queried simulation", "type": "string" }, "throughput": { "description": "Latest effective requests per second", "type": "number" }, "tokensPerSecond": { "description": "Latest inference throughput in tokens/second — present only on GPU inference simulations", "type": "number" }, "traffic": { "description": "Current traffic level in RPS", "type": "number" } }, "type": "object" } }, { "description": "Recover one reversible failed resource in a temporary anonymous demo simulation. No API key required for this temporary anonymous demo operation. Lower traffic to a serviceable level first, then provide resourceId or resourceName from simulation.create, simulation.step, or simulation.metrics. This deactivates applicable instance_down/database_overload failures for only the selected resource and returns recoveryProgress with parked, cooling_down, or healthy state plus cooldown counters. It cannot restore an instance_kill because that failure permanently removes the resource. The likely next tool is simulation.step; keep stepping and inspect the targeted resource until recoveryProgress.state is healthy. Pass simulationId from simulation.create when using a fresh MCP session; a preserved session may omit it. Authenticate with an API key to unlock all 63 tools and unlimited simulations.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "resourceId": { "description": "ID of the failed resource to recover", "type": "string" }, "resourceName": { "description": "Exact case-insensitive name of the failed resource to recover", "type": "string" }, "simulationId": { "description": "Simulation ID returned by simulation.create. Preserve Mcp-Session-Id to omit this field and use the session's current simulation; if your connector starts a fresh MCP session for each call (for example Grok Bot or Cursor), pass this explicit ID after every fresh initialization. A fresh session has no current-simulation pointer and returns NO_ACTIVE_SIMULATION when the ID is omitted. Anonymous capabilities are short-lived (30 minutes by default), unguessable, and revoked when the demo expires or is deleted; proxy IP changes do not invalidate them. Do not treat the ID as a durable share link.", "type": "string" } }, "type": "object" }, "name": "simulation.recover_resource", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "properties": { "deactivatedFailureIds": { "items": { "type": "string" }, "type": "array" }, "previousHealth": { "type": "string" }, "recoveryProgress": { "additionalProperties": false, "properties": { "cooldown": { "additionalProperties": false, "properties": { "completedSteps": { "type": "number" }, "remainingSteps": { "type": "number" }, "requiredSteps": { "type": "number" }, "target": { "anyOf": [ { "enum": [ "warning", "healthy" ], "type": "string" }, { "type": "null" } ] } }, "required": [ "target", "completedSteps", "requiredSteps", "remainingSteps" ], "type": "object" }, "parkWindow": { "additionalProperties": false, "properties": { "completedSteps": { "type": "number" }, "remainingSteps": { "type": "number" }, "totalSteps": { "type": "number" } }, "required": [ "totalSteps", "completedSteps", "remainingSteps" ], "type": "object" }, "state": { "enum": [ "parked", "blocked", "cooling_down", "healthy" ], "type": "string" } }, "required": [ "state", "parkWindow", "cooldown" ], "type": "object" }, "recoveryState": { "type": "string" }, "resolvedResourceId": { "type": "string" }, "resolvedResourceName": { "type": "string" }, "simulationId": { "type": "string" }, "simulationIdSource": { "enum": [ "explicit", "session_default" ], "type": "string" }, "stepsToHealthy": { "type": "number" }, "stepsToHealthyIsLowerBound": { "type": "boolean" } }, "type": "object" } }, { "description": "Advance a temporary anonymous demo simulation by one time step and return updated metrics — CPU, latency, throughput, error rate, cost (max 20 persisted steps per demo). Concurrent simulation.step calls on one simulation either serialize as distinct consecutive steps (in the browser Workspace queue) or receive HTTP 409 simulation_step_in_progress without advancing or consuming a demo credit. Wait for the running call to finish, then retry only rejected calls; distinct simulations can run concurrently. After a timeout, inspect simulation.metrics before retrying an uncertain result. Use it to observe how the architecture behaves over time, typically right after simulation.create or simulation.inject_traffic. Do not use it to read current state without advancing time — that is simulation.metrics. Pass the simulationId returned by simulation.create when your connector opens a fresh MCP session; preserve Mcp-Session-Id to use the omitted-ID current-simulation default. The likely next tool is simulation.step again (to keep observing) or simulation.inject_traffic (to change load first). Responses are compact by default: principal metrics plus per-resource status (id, name, status, cpuPercent, routedRps, availabilityState, isRoutable, and recoveryBlockedReason when provided) and this step's events. Seeded characteristics.eksSpotInterruption telemetry retains its additive migrationEvaluation beside the interruption lifecycle: use its recorded/derived/unavailable field provenance, frozen deadline verdict/counts/reasons, and simulation-clock milestones rather than final service health. The distinct eksSpotMigration contract remains separately reported when configured. Compact responses also include errorBreakdown when the engine provides it. A critical resource with isRoutable: true is degraded but still serving; availabilityState: unavailable and isRoutable: false identify a failed or parked node. Pass responseMode: 'full' to get the complete simulation state instead. During recovery, each resource may include recoveryProgress with state parked, cooling_down, or healthy, plus parkWindow and cooldown counters. Poll simulation.step until the targeted resource's recoveryProgress.state is healthy, then use simulation.metrics to inspect the resulting state and metrics. GPU / inference workflow: when the simulation includes a kubernetes resource with characteristics.inferenceMode: true, each step response also includes gpuUtilization (%), tokensPerSecond, costPerMillionTokens (USD/M tokens), idleGpuCostPerHour (USD/hr of standby GPU spend), and idleGpuFraction (0-1 idle HA overhead share) so you can track inference economics step by step. Authenticate with an API key for unlimited steps and GPU right-sizing hints.", "inputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": false, "properties": { "responseMode": { "default": "compact", "description": "Response detail level. 'compact' (default) returns only principal metrics, errorBreakdown when available, per-resource status (id, name, status, cpuPercent, routedRps, availabilityState, isRoutable, recoveryBlockedReason, and failureLifecycle/routingState when provided), and this step's events — keeps observations small for agent loops. 'full' returns the complete backend step response including the entire simulation object with all resource characteristics and connections.", "enum": [ "compact", "full" ], "type": "string" }, "simulationId": { "description": "Simulation ID returned by simulation.create. Preserve Mcp-Session-Id to omit this field and use the session's current simulation; if your connector starts a fresh MCP session for each call (for example Grok Bot or Cursor), pass this explicit ID after every fresh initialization. A fresh session has no current-simulation pointer and returns NO_ACTIVE_SIMULATION when the ID is omitted. Anonymous capabilities are short-lived (30 minutes by default), unguessable, and revoked when the demo expires or is deleted; proxy IP changes do not invalidate them. Do not treat the ID as a durable share link.", "type": "string" } }, "type": "object" }, "name": "simulation.step", "outputSchema": { "$schema": "http://json-schema.org/draft-07/schema#", "additionalProperties": true, "description": "Compact step summary by default (principal metrics + per-resource status), the complete backend step response with responseMode 'full', or a structured limit response (status: limit_reached) when the step cap is reached", "properties": { "appWeight": { "enum": [ "lean", "typical", "heavy" ], "type": "string" }, "appWeightDefaulted": { "type": "boolean" }, "auroraFailovers": { "description": "Compact per-writer Aurora failover state; promoting has no serving writer, serving means the explicitly related standby took over.", "items": { "additionalProperties": false, "properties": { "failedResourceId": { "type": "string" }, "phase": { "enum": [ "promoting", "serving", "unavailable" ], "type": "string" }, "standbyResourceId": { "type": "string" }, "succeeded": { "type": "boolean" } }, "required": [ "failedResourceId", "standbyResourceId", "succeeded", "phase" ], "type": "object" }, "type": "array" }, "calibrationEvidence": { "additionalProperties": false, "description": "Owned-versus-modeled evidence and latency boundary.", "properties": { "calibrationId": { "description": "Versioned owned calibration identifier; present only when that calibration applies.", "type": "string" }, "kind": { "description": "owned for the exact AWS CRUD fit; modeled when the gate fails or another generic model applies.", "enum": [ "owned", "modeled" ], "type": "string" }, "latencyBasis": { "description": "General modeled latency path, such as in-VPC ALB rather than end-to-end; see latencyP99Basis for percentile-specific P99 provenance.", "type": "string" }, "latencyP99Basis": { "description": "Percentile-specific P99 basis: owned in-VPC internal-ALB fit, scaled from that fit (not directly measured), or uncalibrated generic model.", "type": "string" }, "note": { "description": "Names the fit scope and limitations. For canonical workload inference, states it is a modeling assumption. For modeled fallback, names all actual failed checks in stable order with resource IDs and safe expected/actual values; never only a generic 'not applied'.", "type": "string" } }, "required": [ "kind", "latencyBasis", "note" ], "type": "object" }, "costPerHour": { "description": "Estimated cost in USD/hr", "type": "number" }, "costPerMillionTokens": { "description": "Self-hosted inference cost in USD per million tokens (null when no tokens are being processed) — present only on GPU inference simulations", "type": [ "number", "null" ] }, "currentStep": { "description": "New simulation time step index", "type": "number" }, "effectiveConfigHash": { "description": "Versioned prediction hash over replay startup inputs, engine version, and calibration identity; see replayIdentity.effectiveConfigHash for the original replay-only hash.", "maxLength": 64, "minLength": 64, "type": "string" }, "eksSpotInterruptions": { "description": "Seeded EKS interruption telemetry, including the authoritative additive migrationEvaluation when recorded.", "items": { "additionalProperties": true, "properties": { "checkpointEvidence": { "additionalProperties": false, "description": "Persisted seeded-EKS checkpoint captured with its interruption JSONB record. engineInputStepIndex is an engine input-step index, simulationSeconds is derived from that index and the recorded tick, and traffic provenance points to outer metric fields rather than duplicating float values.", "properties": { "engineInputStepIndex": { "minimum": 0, "type": "integer" }, "fieldProvenance": { "additionalProperties": false, "properties": { "engineInputStepIndex": { "$ref": "#/properties/goodputProvenance" }, "intervalAttribution": { "$ref": "#/properties/goodputProvenance" }, "pointRateSemantics": { "$ref": "#/properties/goodputProvenance" }, "simulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "tickDurationSeconds": { "$ref": "#/properties/goodputProvenance" } }, "required": [ "engineInputStepIndex", "simulationSeconds", "tickDurationSeconds", "pointRateSemantics", "intervalAttribution" ], "type": "object" }, "intervalAttribution": { "anyOf": [ { "anyOf": [ { "additionalProperties": false, "properties": { "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" }, "source": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "source" ], "type": "object" }, "status": { "const": "recorded", "type": "string" }, "window": { "$ref": "#/properties/goodputWindow/anyOf/1/properties/window" } }, "required": [ "status", "window", "provenance" ], "type": "object" }, { "additionalProperties": false, "properties": { "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" }, "status": { "const": "unavailable", "type": "string" } }, "required": [ "status", "provenance" ], "type": "object" } ] }, { "type": "null" } ] }, "pointRateSemantics": { "const": "post_step_point_rate", "type": "string" }, "simulationSeconds": { "minimum": 0, "type": "number" }, "tickDurationSeconds": { "exclusiveMinimum": 0, "type": "number" }, "trafficFieldProvenance": { "additionalProperties": false, "properties": { "errorRatePercent": { "$ref": "#/properties/goodputProvenance" }, "latencyP50Ms": { "$ref": "#/properties/goodputProvenance" }, "offeredRps": { "$ref": "#/properties/goodputProvenance" }, "throughputRps": { "$ref": "#/properties/goodputProvenance" } }, "required": [ "throughputRps", "offeredRps", "errorRatePercent", "latencyP50Ms" ], "type": "object" } }, "required": [ "engineInputStepIndex", "simulationSeconds", "tickDurationSeconds", "pointRateSemantics", "intervalAttribution", "fieldProvenance", "trafficFieldProvenance" ], "type": "object" }, "migrationEvaluation": { "additionalProperties": false, "description": "Authoritative additive seeded EKS interruption evaluation. Field values carry recorded/derived/unavailable provenance; deadline verdict/counts/reasons are frozen once status is completed and do not describe later service health.", "properties": { "affectedWorkloadCount": { "anyOf": [ { "minimum": 0, "type": "integer" }, { "type": "null" } ] }, "allAffectedWorkloadsReadyAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "deadlineAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "deadlineMet": { "type": [ "boolean", "null" ] }, "deadlineSeconds": { "const": 120, "type": "number" }, "evaluatedAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "fieldProvenance": { "additionalProperties": false, "properties": { "affectedWorkloadCount": { "$ref": "#/properties/goodputProvenance" }, "allAffectedWorkloadsReadyAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "deadlineAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "deadlineMet": { "$ref": "#/properties/goodputProvenance" }, "deadlineSeconds": { "$ref": "#/properties/goodputProvenance" }, "evaluatedAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "interruptionNoticeAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "limitingReasons": { "$ref": "#/properties/goodputProvenance" }, "migrationDurationSeconds": { "$ref": "#/properties/goodputProvenance" }, "migrationStartedAtSimulationSeconds": { "$ref": "#/properties/goodputProvenance" }, "missedDeadlineWorkloadCount": { "$ref": "#/properties/goodputProvenance" }, "readyWorkloadCountAtDeadline": { "$ref": "#/properties/goodputProvenance" }, "status": { "$ref": "#/properties/goodputProvenance" } }, "required": [ "status", "interruptionNoticeAtSimulationSeconds", "deadlineAtSimulationSeconds", "migrationStartedAtSimulationSeconds", "allAffectedWorkloadsReadyAtSimulationSeconds", "migrationDurationSeconds", "deadlineSeconds", "deadlineMet", "affectedWorkloadCount", "readyWorkloadCountAtDeadline", "missedDeadlineWorkloadCount", "evaluatedAtSimulationSeconds", "limitingReasons" ], "type": "object" }, "interruptionNoticeAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "limitingReasons": { "items": { "additionalProperties": false, "properties": { "code": { "enum": [ "no_reschedule", "scheduling_capacity_exhausted", "image_pull_incomplete", "startup_incomplete", "unclassified_runtime_work_remaining" ], "type": "string" }, "interruptionHandling": { "enum": [ "reschedule", "drain-only", "fail-fast" ], "type": "string" }, "pendingWorkloadCount": { "minimum": 0, "type": "integer" }, "provenance": { "const": "recorded", "type": "string" }, "pullingWorkloadCount": { "minimum": 0, "type": "integer" }, "residualPullSeconds": { "minimum": 0, "type": "number" }, "residualStartupSeconds": { "minimum": 0, "type": "number" }, "schedulingCapacity": { "minimum": 0, "type": "integer" }, "startingWorkloadCount": { "minimum": 0, "type": "integer" }, "workloadCount": { "minimum": 0, "type": "integer" } }, "required": [ "code", "workloadCount", "pendingWorkloadCount", "pullingWorkloadCount", "startingWorkloadCount", "residualPullSeconds", "residualStartupSeconds", "schedulingCapacity", "interruptionHandling", "provenance" ], "type": "object" }, "maxItems": 5, "type": "array" }, "migrationDurationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "migrationStartedAtSimulationSeconds": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ] }, "missedDeadlineWorkloadCount": { "anyOf": [ { "minimum": 0, "type": "integer" }, { "type": "null" } ] }, "readyWorkloadCountAtDeadline": { "anyOf": [ { "minimum": 0, "type": "integer" }, { "type": "null" } ] }, "status": { "enum": [ "not_started", "in_progress", "completed" ], "type": "string" } }, "required": [ "status", "interruptionNoticeAtSimulationSeconds", "deadlineAtSimulationSeconds", "migrationStartedAtSimulationSeconds", "allAffectedWorkloadsReadyAtSimulationSeconds", "migrationDurationSeconds", "deadlineSeconds", "deadlineMet", "affectedWorkloadCount", "readyWorkloadCountAtDeadline", "missedDeadlineWorkloadCount", "evaluatedAtSimulationSeconds", "limitingReasons", "fieldProvenance" ], "type": "object" }, "name": { "type": "string" }, "resourceId": { "type": "string" } }, "type": "object" }, "type": "array" }, "engineVersion": { "description": "Simulation engine version used for this prediction.", "type": "string" }, "errorBreakdown": { "additionalProperties": false, "description": "Validated additive error contributors in percentage-point units; separates pool/DB, compute, capacity, CPU, storage, runtime-memory, and queue absorption effects", "properties": { "capacityOverload": { "type": "number" }, "computeFailure": { "type": "number" }, "cpuOverload": { "type": "number" }, "dbFailure": { "type": "number" }, "dependencyFailure": { "type": "number" }, "ociStorage": { "type": "number" }, "poolSaturation": { "type": "number" }, "queueAbsorption": { "type": "number" }, "runtimeMemory": { "type": "number" } }, "required": [ "poolSaturation", "dbFailure", "computeFailure", "capacityOverload", "cpuOverload", "ociStorage", "queueAbsorption" ], "type": "object" }, "errorRate": { "description": "Client-level error rate (%) against original offered RPS. Resilience dependency targets in autoscaled compute fleets use aggregate routable-fleet capacity and health; non-autoscaled targets use their resolved resource capacity, including any declared fixed-instance count.", "type": "number" }, "events": { "description": "Events generated during this step", "items": { "additionalProperties": {}, "type": "object" }, "type": "array" }, "goodputProvenance": { "anyOf": [ { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" } }, "required": [ "kind" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "derived", "type": "string" }, "sourceFields": { "items": { "minLength": 1, "type": "string" }, "minItems": 1, "type": "array" } }, "required": [ "kind", "sourceFields" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" } ], "description": "Provenance for the modeled goodput field" }, "goodputRps": { "description": "Modeled client-level successful requests per second; a post-step point rate sourced from metrics.throughput. Resilience dependency targets in autoscaled compute fleets use aggregate routable-fleet capacity and health; non-autoscaled targets use their resolved resource capacity, including any declared fixed-instance count.", "type": "number" }, "goodputSemantics": { "const": "post_step_point_rate", "description": "Goodput is a point rate, not an interval total", "type": "string" }, "goodputWindow": { "anyOf": [ { "additionalProperties": false, "properties": { "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" }, "status": { "const": "unavailable", "type": "string" } }, "required": [ "status", "provenance" ], "type": "object" }, { "additionalProperties": false, "properties": { "aggregate": { "additionalProperties": false, "properties": { "coveredDurationSeconds": { "minimum": 0, "type": "number" }, "goodputRequestTotal": { "minimum": 0, "type": "number" }, "goodputRps": { "minimum": 0, "type": "number" }, "provenance": { "anyOf": [ { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" }, "source": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "source" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "derived", "type": "string" }, "sourceFields": { "items": { "maxLength": 256, "minLength": 1, "type": "string" }, "maxItems": 256, "minItems": 1, "type": "array" } }, "required": [ "kind", "sourceFields" ], "type": "object" }, { "additionalProperties": false, "properties": { "kind": { "const": "unavailable", "type": "string" }, "reason": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "reason" ], "type": "object" } ] }, "sampleIds": { "items": { "minLength": 1, "type": "string" }, "type": "array" }, "window": { "$ref": "#/properties/goodputWindow/anyOf/1/properties/window" } }, "required": [ "window", "coveredDurationSeconds", "goodputRps", "goodputRequestTotal", "sampleIds", "provenance" ], "type": "object" }, "provenance": { "additionalProperties": false, "properties": { "kind": { "const": "recorded", "type": "string" }, "source": { "maxLength": 256, "minLength": 1, "type": "string" } }, "required": [ "kind", "source" ], "type": "object" }, "status": { "const": "recorded", "type": "string" }, "window": { "additionalProperties": false, "properties": { "inclusion": { "const": "[start,end)", "type": "string" }, "windowEndSeconds": { "minimum": 0, "type": "number" }, "windowStartSeconds": { "minimum": 0, "type": "number" } }, "required": [ "windowStartSeconds", "windowEndSeconds", "inclusion" ], "type": "object" } }, "required": [ "status", "window", "aggregate", "provenance" ], "type": "object" } ], "description": "Interval goodput aggregate when every point has persisted simulation-clock bounds; otherwise status=unavailable. Never derive this from retrieval time or currentStep." }, "gpuUtilization": { "description": "GPU utilization (%) — present only on simulations with a GPU inference kubernetes resource", "type": "number" }, "idleGpuCostPerHour": { "description": "USD/hr of GPU spend funding idle/standby capacity (HA overhead) — present only on GPU inference simulations", "type": "number" }, "idleGpuFraction": { "description": "Share (0-1) of the GPU bill that is idle/standby capacity — present only on GPU inference simulations; values above 0.5 mean over half the GPU spend is HA overhead", "type": "number" }, "latencyAvailability": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyAvailability" }, "latencyBasis": { "description": "General modeled latency path/boundary; see latencyP99Basis for P99-specific provenance.", "type": "string" }, "latencyP50": { "description": "50th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP95": { "description": "95th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP99": { "description": "Modeled 99th-percentile latency in ms; null when workload-specific latency evidence is unavailable. Interpret with latencyP99Basis and predictionEvidence.latencyP99.", "type": [ "number", "null" ] }, "latencyP99Basis": { "description": "Percentile-specific P99 basis: owned fit, scaled-from-fit (not directly measured), or uncalibrated generic model.", "type": "string" }, "metricId": { "description": "Storage-assigned persisted metric ID when available", "type": "string" }, "metrics": { "additionalProperties": true, "description": "Full-mode backend metric record; migrationEvaluation is exact-typed when a seeded interruption is present.", "properties": { "eksSpotInterruptions": { "items": { "$ref": "#/properties/eksSpotInterruptions/items" }, "type": "array" }, "latencyBasis": { "description": "General modeled latency path/boundary; see latencyP99Basis for P99-specific provenance.", "type": "string" }, "latencyP50": { "description": "50th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP95": { "description": "95th-percentile latency in ms; null when workload-specific latency evidence is unavailable.", "type": [ "number", "null" ] }, "latencyP99": { "description": "Modeled 99th-percentile latency in ms; null when workload-specific latency evidence is unavailable. Interpret with this metric's latencyP99Basis and predictionEvidence.latencyP99.", "type": [ "number", "null" ] }, "latencyP99Basis": { "description": "Percentile-specific P99 basis for this metric checkpoint.", "type": "string" } }, "type": "object" }, "modeledShedRps": { "description": "Aggregate modeled requests per second shed by bounded capacity", "type": "number" }, "offeredRps": { "description": "Aggregate offered requests per second represented by metrics.offeredRps provenance", "type": "number" }, "predictionEffectiveConfigHash": { "description": "Versioned prediction hash over replay startup inputs, engine version, and calibration identity.", "maxLength": 64, "minLength": 64, "type": "string" }, "predictionEvidence": { "anyOf": [ { "additionalProperties": false, "properties": { "appCpu": { "anyOf": [ { "additionalProperties": false, "properties": { "central": { "minimum": 0, "type": "number" }, "high": { "minimum": 0, "type": "number" }, "low": { "minimum": 0, "type": "number" } }, "required": [ "low", "central", "high" ], "type": "object" }, { "type": "null" } ] }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "legacyGeneric": { "type": "boolean" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId", "legacyGeneric" ], "type": "object" }, "type": "array" }, "appWeight": { "enum": [ "lean", "typical", "heavy" ], "type": "string" }, "appWeightDefaulted": { "type": "boolean" }, "evidenceLevel": { "enum": [ "measured", "scaled from measured", "reference estimate" ], "type": "string" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyAvailability": { "additionalProperties": false, "properties": { "reason": { "minLength": 1, "type": "string" }, "resourceIds": { "items": { "minLength": 1, "type": "string" }, "type": "array" }, "status": { "const": "unavailable", "type": "string" } }, "required": [ "status", "reason", "resourceIds" ], "type": "object" }, "latencyP50": { "anyOf": [ { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" } }, "required": [ "low", "central", "high", "evidenceLevel" ], "type": "object" }, { "type": "null" } ] }, "latencyP95": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP99": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyPercentileStatus": { "enum": [ "modeled", "unavailable" ], "type": "string" }, "legacyGeneric": { "type": "boolean" }, "loadScope": { "enum": [ "within measured load", "beyond measured load", "below measured load", "not calibrated" ], "type": "string" }, "note": { "type": "string" }, "requestServingCapacity": { "items": { "additionalProperties": false, "properties": { "aggregateCapacityRps": { "minimum": 0, "type": "number" }, "capacityEvidence": { "additionalProperties": false, "properties": { "basis": { "enum": [ "assumption", "documented", "measured", "catalog" ], "type": "string" }, "origin": { "enum": [ "caller-provided", "fargate-size-heuristic", "policy-default", "provider-catalog" ], "type": "string" }, "source": { "type": "string" } }, "required": [ "basis", "origin" ], "type": "object" }, "maxAggregateCapacityRps": { "minimum": 0, "type": "number" }, "maxTaskCount": { "exclusiveMinimum": 0, "type": "integer" }, "perTaskCapacityRps": { "exclusiveMinimum": 0, "type": "number" }, "resourceId": { "type": "string" }, "resourceName": { "type": "string" }, "taskCount": { "minimum": 0, "type": "integer" } }, "required": [ "resourceId", "resourceName", "taskCount", "perTaskCapacityRps", "aggregateCapacityRps", "maxTaskCount", "maxAggregateCapacityRps", "capacityEvidence" ], "type": "object" }, "type": "array" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "legacyGeneric", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" }, { "additionalProperties": false, "properties": { "appCpu": { "type": "null" }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId" ], "type": "object" }, "maxItems": 0, "type": "array" }, "appWeight": { "type": "null" }, "appWeightDefaulted": { "type": "null" }, "evidenceLevel": { "const": "unavailable/legacy", "type": "string" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyP50": { "type": "null" }, "latencyP95": { "type": "null" }, "latencyP99": { "type": "null" }, "legacyGeneric": { "type": "null" }, "loadScope": { "type": "null" }, "note": { "type": "string" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" }, { "additionalProperties": false, "properties": { "appCpu": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0" }, { "type": "null" } ] }, "appCpuByResource": { "items": { "additionalProperties": false, "properties": { "central": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/central" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "high": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/high" }, "legacyGeneric": { "type": "boolean" }, "low": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appCpu/anyOf/0/properties/low" }, "resourceId": { "type": "string" } }, "required": [ "low", "central", "high", "evidenceLevel", "resourceId" ], "type": "object" }, "type": "array" }, "appWeight": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/appWeight" }, "appWeightDefaulted": { "type": "boolean" }, "evidenceLevel": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/evidenceLevel" }, "formulaIds": { "items": { "type": "string" }, "type": "array" }, "latencyAvailability": { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyAvailability" }, "latencyP50": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP95": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyP99": { "anyOf": [ { "$ref": "#/properties/predictionEvidence/anyOf/0/properties/latencyP50/anyOf/0" }, { "type": "null" } ] }, "latencyPercentileStatus": { "enum": [ "modeled", "unavailable" ], "type": "string" }, "legacyGeneric": { "type": "boolean" }, "loadScope": { "enum": [ "within measured load", "beyond measured load", "below measured load", "not calibrated" ], "type": "string" }, "note": { "type": "string" }, "sourceIds": { "items": { "type": "string" }, "type": "array" }, "version": { "const": 1, "type": "number" } }, "required": [ "version", "evidenceLevel", "appWeight", "appWeightDefaulted", "loadScope", "sourceIds", "formulaIds", "appCpu", "appCpuByResource", "latencyP50", "latencyP95", "note" ], "type": "object" } ] }, "replayIdentity": { "additionalProperties": false, "properties": { "effectiveConfigHash": { "description": "Original replay-only SHA-256 of the effective six-control startup configuration; top-level effectiveConfigHash is the versioned prediction identity.", "maxLength": 64, "minLength": 64, "type": "string" }, "scenarioHash": { "description": "Canonical SHA-256 of the persisted scenario graph and attached traffic-pattern order", "maxLength": 64, "minLength": 64, "type": "string" } }, "required": [ "scenarioHash", "effectiveConfigHash" ], "type": "object" }, "resilienceDiagnostics": { "additionalProperties": true, "description": "Bounded resilience diagnostics summary (compact mode). Absent when the resilience model did not run. Use simulation.compare_resilience for full per-path detail.", "properties": { "bounded": { "description": "True when a generated-traffic, traversal-work, or cascade-depth bound truncated model work", "type": "boolean" }, "capacityByResource": { "description": "Unique logical dependency resources; each appears once even when several paths reference it. When values are finite, nominalCapacityRps minus warmupCapacityLossRps (members still warming) and healthCapacityLossRps (unavailable or degraded members) equals routableCapacityRps, within rounding.", "items": { "additionalProperties": false, "properties": { "healthCapacityLossRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity lost because members are unavailable or degraded; zero means no modeled member-health loss" }, "nominalCapacityRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity before warm-up and member-health reductions; null means no finite capacity ceiling is known" }, "resourceId": { "description": "Logical resource ID used by one or more resilience dependency paths", "type": "string" }, "role": { "description": "Whether this resource is a dependency source, target, or both", "enum": [ "source", "target", "source_and_target" ], "type": "string" }, "routableCapacityRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity available to this logical resource after modeled warm-up and member health; null means no finite capacity ceiling is known" }, "warmupCapacityLossRps": { "anyOf": [ { "minimum": 0, "type": "number" }, { "type": "null" } ], "description": "Aggregate capacity lost because fleet members are still warming up; zero means no modeled warm-up loss" } }, "required": [ "resourceId", "role", "routableCapacityRps" ], "type": "object" }, "type": "array" }, "incidentOutcome": { "description": "Incident outcome: stable | degraded | cascading | protected | recovered", "type": "string" }, "pathCount": { "description": "Number of dependency paths evaluated this step", "type": "number" } }, "type": "object" }, "resources": { "description": "Per-resource status summary (compact mode)", "items": { "additionalProperties": true, "properties": { "availabilityState": { "description": "Availability derived from routed traffic: available = serving normally, degraded = critical/warning but still serving, unavailable = no traffic and not routable", "enum": [ "available", "degraded", "unavailable" ], "type": "string" }, "cpuPercent": { "description": "CPU utilization (%)", "type": "number" }, "failureLifecycle": { "description": "Read-only failure lifecycle marker when the backend identifies a quick injection or typed instance failure", "enum": [ "quick_injection_parked", "quick_injection_rejoined", "instance_down", "instance_kill" ], "type": "string" }, "id": { "description": "Resource ID", "type": "string" }, "isRoutable": { "description": "Whether this compute/Kubernetes resource can receive traffic in this step; false distinguishes a failed/parked node from a critical node still serving", "type": "boolean" }, "name": { "description": "Resource display name", "type": "string" }, "recoveryBlockedReason": { "description": "Engine recovery guard currently preventing cooldown progress, when present (for example failure_park_window or idle_cpu_floor)", "type": "string" }, "recoveryProgress": { "additionalProperties": false, "description": "Read-only recovery progress; poll simulation.step until state is healthy, then use simulation.metrics to inspect the result", "properties": { "cooldown": { "additionalProperties": false, "properties": { "completedSteps": { "type": "number" }, "remainingSteps": { "type": "number" }, "requiredSteps": { "type": "number" }, "target": { "anyOf": [ { "enum": [ "warning", "healthy" ], "type": "string" }, { "type": "null" } ] } }, "required": [ "target", "completedSteps", "requiredSteps", "remainingSteps" ], "type": "object" }, "parkWindow": { "additionalProperties": false, "properties": { "completedSteps": { "type": "number" }, "remainingSteps": { "type": "number" }, "totalSteps": { "type": "number" } }, "required": [ "totalSteps", "completedSteps", "remainingSteps" ], "type": "object" }, "state": { "enum": [ "parked", "blocked", "cooling_down", "healthy" ], "type": "string" } }, "required": [ "state", "parkWindow", "cooldown" ], "type": "object" }, "routedRps": { "description": "Requests per second routed to this resource this step (compute/kubernetes only)", "type": "number" }, "routingState": { "description": "Read-only current routing state for a resource with failure lifecycle telemetry", "enum": [ "unavailable", "serving" ], "type": "string" }, "status": { "description": "Health status (healthy/warning/critical/failed)", "type": "string" } }, "type": "object" }, "type": "array" }, "retryAmplificationFactor": { "description": "(Original offered RPS + generated retry RPS) / original offered RPS; an amplification measure, not capacity or goodput. Values > 1.0 = amplification risk. null = model ran but no traffic. Absent = resilience model disabled.", "type": [ "number", "null" ] }, "scenarioHash": { "description": "Canonical SHA-256 of the persisted scenario graph and attached traffic-pattern order", "maxLength": 64, "minLength": 64, "type": "string" }, "simulationId": { "description": "ID of the stepped simulation", "type": "string" }, "throughput": { "description": "Effective requests per second", "type": "number" }, "tokensPerSecond": { "description": "Inference throughput in tokens/second — present only on GPU inference simulations", "type": "number" }, "traffic": { "description": "Current traffic level in RPS", "type": "number" } }, "type": "object" } } ] }
Verify it yourselfcurl -s https://api.teppi.xyz/v1/evidence/sha256:62b4ac86a8b9949636fbf8820cc651f16e3a0a955127cd423596b81a7d866129 | sha256sum