Endpoints: 28,729MCP servers: 18,413Payout addresses: 2,071Paid calls: 1,545Letters: 14Defects: 1,324counted 1 min ago
teppi

Server definition

Hash
sha256:10d0b907384dfe0996ca4740daff77241cb56178fb796ede260c19c4f56f7b0a
What it is
What a remote MCP server returned when asked what it offers: 6 tools

The blob, as servednamed by its sha256

{ "instructions": "Answers whether an open-weight LLM fits a GPU, with the formulas nodegrove.io publishes. Use can_i_run for \"can my GPU run this model\", what_fits for \"what can my GPU run\", estimate_vram for memory at each quantisation, estimate_from_hf_repo to read any Hugging Face repo, and list_models / list_gpus to find ids. Figures are estimates from stated formulas over config.json values and makers' specs, never benchmarks, and speeds are upper bounds: say so when you quote them, and give the page link from the result. The data is CC BY 4.0: credit Nodegrove (nodegrove.io).", "tools": [ { "description": "Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context that fits and, on a no, every change that would make it fit: quantisation, KV cache, context, another card or a smaller model. Model: a name or id from list_models, any Hugging Face repo id, or its architecture. GPU: a name or id from list_gpus, or vram_gb for any other card.", "inputSchema": { "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { "active_params_b": { "description": "Parameters read per token, billions, for a mixture-of-experts model read from Hugging Face (from its model card). Sets the speed ceiling.", "exclusiveMinimum": 0, "maximum": 10000, "type": "number" }, "apple_silicon": { "description": "vram_gb is Apple unified memory; the GPU can use about 75% of it by default.", "type": "boolean" }, "architecture": { "description": "A model described by its config.json values instead of a name.", "properties": { "active_params_b": { "description": "Parameters read per token, billions (mixture-of-experts only).", "exclusiveMinimum": 0, "maximum": 10000, "type": "number" }, "fixed_state_gb": { "description": "Fixed recurrent state of linear-attention or Mamba layers, GB.", "maximum": 100, "minimum": 0, "type": "number" }, "head_dim": { "description": "head_dim, or hidden_size ÷ num_attention_heads", "exclusiveMinimum": 0, "maximum": 4096, "type": "integer" }, "kv_groups": { "description": "Only for non-standard attention: one entry per group of layers that cache the same way. Replaces layers × kv_heads × head_dim.", "items": { "properties": { "layers": { "exclusiveMinimum": 0, "maximum": 9007199254740991, "type": "integer" }, "values_per_token": { "description": "Values each layer caches per token: 2 × KV heads × head dim, or the latent width for MLA.", "exclusiveMinimum": 0, "type": "number" }, "window_tokens": { "description": "Sliding window: these layers keep only this many tokens.", "exclusiveMinimum": 0, "maximum": 9007199254740991, "type": "integer" } }, "required": [ "layers", "values_per_token" ], "type": "object" }, "maxItems": 8, "type": "array" }, "kv_heads": { "description": "num_key_value_heads", "exclusiveMinimum": 0, "maximum": 1024, "type": "integer" }, "layers": { "description": "num_hidden_layers", "exclusiveMinimum": 0, "maximum": 1000, "type": "integer" }, "native_context": { "description": "The context window the model supports, tokens.", "exclusiveMinimum": 0, "maximum": 9007199254740991, "type": "integer" }, "params_b": { "description": "Total parameters, billions; all experts for a mixture-of-experts model.", "exclusiveMinimum": 0, "maximum": 10000, "type": "number" } }, "required": [ "params_b", "layers", "kv_heads", "head_dim" ], "type": "object" }, "bandwidth_gb_s": { "description": "Memory bandwidth from the maker's spec, GB/s, for a speed ceiling.", "exclusiveMinimum": 0, "maximum": 100000, "type": "number" }, "context": { "default": 8192, "description": "Tokens held in context: prompt plus conversation.", "maximum": 10000000, "minimum": 1, "type": "integer" }, "gpu": { "description": "A GPU from list_gpus (id or name, e.g. \"rtx-4090\", \"4090\" or \"M4 Max\").", "maxLength": 100, "minLength": 1, "type": "string" }, "kv_cache": { "default": "fp16", "description": "KV cache precision. fp16 is what most runtimes use; q8 halves the cache.", "enum": [ "fp16", "q8" ], "type": "string" }, "model": { "description": "A model from list_models (id or name, e.g. \"llama-3.3-70b\" or \"Llama 3.3 70B\"), or any Hugging Face repo id (e.g. \"Qwen/Qwen3-8B\"), read live from its config.json.", "maxLength": 200, "minLength": 1, "type": "string" }, "quant": { "default": "q4", "description": "Weight quantisation: fp16 (FP16 / BF16), q8 (Q8_0), q6 (Q6_K), q5 (Q5_K_M), q4 (Q4_K_M), q3 (Q3_K_M). q4 is the common default.", "enum": [ "fp16", "q8", "q6", "q5", "q4", "q3" ], "type": "string" }, "vram_gb": { "description": "Memory of a card not in list_gpus, GB. For a Mac, its unified memory with apple_silicon: true.", "exclusiveMinimum": 0, "maximum": 4096, "type": "number" } }, "type": "object" }, "name": "can_i_run", "outputSchema": null }, { "description": "Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or latent), how much each 1,000 tokens of context costs, and weights + KV cache + overhead at every quantisation. For models nodegrove.io has not reviewed; anything the reader cannot model is listed in warnings.", "inputSchema": { "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { "active_params_b": { "description": "Parameters read per token, billions, for a mixture-of-experts model read from Hugging Face (from its model card). Sets the speed ceiling.", "exclusiveMinimum": 0, "maximum": 10000, "type": "number" }, "context": { "default": 8192, "description": "Tokens held in context: prompt plus conversation.", "maximum": 10000000, "minimum": 1, "type": "integer" }, "kv_cache": { "default": "fp16", "description": "KV cache precision. fp16 is what most runtimes use; q8 halves the cache.", "enum": [ "fp16", "q8" ], "type": "string" }, "repo": { "description": "Hugging Face repo id, e.g. \"Qwen/Qwen3-8B\", or its huggingface.co URL.", "maxLength": 200, "minLength": 3, "type": "string" } }, "required": [ "repo" ], "type": "object" }, "name": "estimate_from_hf_repo", "outputSchema": null }, { "description": "How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds each. Model: a name or id from list_models, any Hugging Face repo id, or its architecture (params_b, layers, kv_heads, head_dim).", "inputSchema": { "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { "active_params_b": { "description": "Parameters read per token, billions, for a mixture-of-experts model read from Hugging Face (from its model card). Sets the speed ceiling.", "exclusiveMinimum": 0, "maximum": 10000, "type": "number" }, "architecture": { "description": "A model described by its config.json values instead of a name.", "properties": { "active_params_b": { "description": "Parameters read per token, billions (mixture-of-experts only).", "exclusiveMinimum": 0, "maximum": 10000, "type": "number" }, "fixed_state_gb": { "description": "Fixed recurrent state of linear-attention or Mamba layers, GB.", "maximum": 100, "minimum": 0, "type": "number" }, "head_dim": { "description": "head_dim, or hidden_size ÷ num_attention_heads", "exclusiveMinimum": 0, "maximum": 4096, "type": "integer" }, "kv_groups": { "description": "Only for non-standard attention: one entry per group of layers that cache the same way. Replaces layers × kv_heads × head_dim.", "items": { "properties": { "layers": { "exclusiveMinimum": 0, "maximum": 9007199254740991, "type": "integer" }, "values_per_token": { "description": "Values each layer caches per token: 2 × KV heads × head dim, or the latent width for MLA.", "exclusiveMinimum": 0, "type": "number" }, "window_tokens": { "description": "Sliding window: these layers keep only this many tokens.", "exclusiveMinimum": 0, "maximum": 9007199254740991, "type": "integer" } }, "required": [ "layers", "values_per_token" ], "type": "object" }, "maxItems": 8, "type": "array" }, "kv_heads": { "description": "num_key_value_heads", "exclusiveMinimum": 0, "maximum": 1024, "type": "integer" }, "layers": { "description": "num_hidden_layers", "exclusiveMinimum": 0, "maximum": 1000, "type": "integer" }, "native_context": { "description": "The context window the model supports, tokens.", "exclusiveMinimum": 0, "maximum": 9007199254740991, "type": "integer" }, "params_b": { "description": "Total parameters, billions; all experts for a mixture-of-experts model.", "exclusiveMinimum": 0, "maximum": 10000, "type": "number" } }, "required": [ "params_b", "layers", "kv_heads", "head_dim" ], "type": "object" }, "context": { "default": 8192, "description": "Tokens held in context: prompt plus conversation.", "maximum": 10000000, "minimum": 1, "type": "integer" }, "kv_cache": { "default": "fp16", "description": "KV cache precision. fp16 is what most runtimes use; q8 halves the cache.", "enum": [ "fp16", "q8" ], "type": "string" }, "model": { "description": "A model from list_models (id or name, e.g. \"llama-3.3-70b\" or \"Llama 3.3 70B\"), or any Hugging Face repo id (e.g. \"Qwen/Qwen3-8B\"), read live from its config.json.", "maxLength": 200, "minLength": 1, "type": "string" }, "quant": { "description": "Weight quantisation: fp16 (FP16 / BF16), q8 (Q8_0), q6 (Q6_K), q5 (Q5_K_M), q4 (Q4_K_M), q3 (Q3_K_M). Omit it for all 6.", "enum": [ "fp16", "q8", "q6", "q5", "q4", "q3" ], "type": "string" } }, "type": "object" }, "name": "estimate_vram", "outputSchema": null }, { "description": "The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page.", "inputSchema": { "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { "search": { "description": "Words to filter by, e.g. \"qwen\" or \"24 GB\".", "maxLength": 100, "type": "string" } }, "type": "object" }, "name": "list_gpus", "outputSchema": null }, { "description": "The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-09-25): id, size, attention design, native context, licence, memory at Q4 with 8k context and each model's page.", "inputSchema": { "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { "search": { "description": "Words to filter by, e.g. \"qwen\" or \"24 GB\".", "maxLength": 100, "type": "string" } }, "type": "object" }, "name": "list_models", "outputSchema": null }, { "description": "Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class that fits with room for context at conversational speed), the largest that fits, the best at Q8 and the first out of reach. GPU: a name or id from list_gpus, or vram_gb for any other card.", "inputSchema": { "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { "apple_silicon": { "description": "vram_gb is Apple unified memory; the GPU can use about 75% of it by default.", "type": "boolean" }, "bandwidth_gb_s": { "description": "Memory bandwidth from the maker's spec, GB/s, for a speed ceiling.", "exclusiveMinimum": 0, "maximum": 100000, "type": "number" }, "context": { "default": 8192, "description": "Tokens held in context: prompt plus conversation.", "maximum": 10000000, "minimum": 1, "type": "integer" }, "gpu": { "description": "A GPU from list_gpus (id or name, e.g. \"rtx-4090\", \"4090\" or \"M4 Max\").", "maxLength": 100, "minLength": 1, "type": "string" }, "quant": { "default": "q4", "description": "Weight quantisation: fp16 (FP16 / BF16), q8 (Q8_0), q6 (Q6_K), q5 (Q5_K_M), q4 (Q4_K_M), q3 (Q3_K_M). q4 is the common default.", "enum": [ "fp16", "q8", "q6", "q5", "q4", "q3" ], "type": "string" }, "vram_gb": { "description": "Memory of a card not in list_gpus, GB. For a Mac, its unified memory with apple_silicon: true.", "exclusiveMinimum": 0, "maximum": 4096, "type": "number" } }, "type": "object" }, "name": "what_fits", "outputSchema": null } ] }
Verify it yourselfcurl -s https://api.teppi.xyz/v1/evidence/sha256:10d0b907384dfe0996ca4740daff77241cb56178fb796ede260c19c4f56f7b0a | sha256sum