LLM Configurator
Read-only: which local LLMs a GPU or Mac can run - VRAM fit, tokens/sec, model specs.
Should I use this
Quality & Safety
Based on automated analysis of tool definitions and protocol compliance.
Context Cost
This is the approximate number of tokens consumed each time the server's tools are loaded into a model's context. Higher counts reduce the attention available for other tasks.
Install
One-Click Install
Add this to your `claude_desktop_config.json` file:
{
"mcpServers": {
"llmconfigurator": {
"url": "https://elgadmisijrfowuujcmu.supabase.co/functions/v1/mcp"
}
}
}Remote endpoints
https://elgadmisijrfowuujcmu.supabase.co/functions/v1/mcpstreamable-httpWhat it can do
Tool inventory
Tools (5)
🟢search_catalog(query, kind, limit)
Use this first when the user names a GPU, Mac, mini PC or language model and you need its id for the other tools. Matches names and common aliases in a fixed catalogue of local-LLM hardware and models. It does not return specs, fit or speed, and it does not search the web.
Input Schema
{
"type": "object",
"properties": {
"query": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[^\\u0000-\\u001f\\u007f]*$",
"description": "Hardware or model name as the user wrote it, e.g. \"4090\", \"m4 max 64\", \"llama 3.3 70b\"."
},
"kind": {
"type": "string",
"enum": [
"hardware",
"model"
],
"description": "Search hardware or models. Required."
},
"limit": {
"default": 5,
"description": "Maximum matches, 1 to 10. Default 5.",
"type": "integer",
"minimum": 1,
"maximum": 10
}
},
"required": [
"query",
"kind"
]
}Output Schema
{
"type": "object",
"properties": {
"data_as_of": {
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$",
"description": "Date the bundled catalogue was last updated. Not the time of this call."
},
"summary": {
"type": "string",
"maxLength": 600,
"description": "One or two plain-language sentences stating the answer."
},
"kind": {
"type": "string",
"enum": [
"hardware",
"model"
]
},
"matches": {
"type": "array",
"items": {
"type": "object",
"properties": {
"id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$",
"description": "Pass this id to the other tools."
},
"name": {
"type": "string",
"maxLength": 120
},
"kind": {
"type": "string",
"enum": [
"hardware",
"model"
]
},
"detail": {
"type": "string",
"maxLength": 160,
"description": "A short factual line, e.g. \"24 GB, 1008 GB/s\" or \"8B dense\"."
},
"url": {
"type": "string",
"pattern": "^https:\\/\\/llmconfigurator\\.com\\/en\\/"
}
},
"required": [
"id",
"name",
"kind",
"detail",
"url"
],
"additionalProperties": false
}
}
},
"required": [
"data_as_of",
"summary",
"kind",
"matches"
],
"additionalProperties": false
}🟢check_hardware_fit(hardware_id, model_id, quantisation, context_length)
Use this when the user asks whether one specific model runs on one specific machine, at what memory cost, and roughly how fast. Needs a hardware id and a model id from search_catalog. Returns a fit verdict, the memory arithmetic, and a decode-speed range with a confidence label. Memory is sized from all parameters, speed from active parameters. It does not recommend purchases and does not run anything.
Input Schema
{
"type": "object",
"properties": {
"hardware_id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$",
"description": "Hardware id from search_catalog."
},
"model_id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$",
"description": "Model id from search_catalog."
},
"quantisation": {
"default": "Q4_K_M",
"description": "Weight quantisation. One of Q2_K, Q3_K_M, Q4_K_M, Q5_K_M, Q6_K, Q8_0, F16. Default Q4_K_M.",
"type": "string",
"enum": [
"Q2_K",
"Q3_K_M",
"Q4_K_M",
"Q5_K_M",
"Q6_K",
"Q8_0",
"F16"
]
},
"context_length": {
"default": 4096,
"description": "Context window in tokens that the KV cache is sized for. 512 to 262144. Default 4096.",
"type": "integer",
"minimum": 512,
"maximum": 262144
}
},
"required": [
"hardware_id",
"model_id"
]
}Output Schema
{
"type": "object",
"properties": {
"data_as_of": {
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$",
"description": "Date the bundled catalogue was last updated. Not the time of this call."
},
"summary": {
"type": "string",
"maxLength": 600,
"description": "One or two plain-language sentences stating the answer."
},
"verdict": {
"type": "string",
"enum": [
"fits_comfortably",
"fits_tight",
"needs_offload",
"does_not_fit"
]
},
"quantisation": {
"type": "string",
"enum": [
"Q2_K",
"Q3_K_M",
"Q4_K_M",
"Q5_K_M",
"Q6_K",
"Q8_0",
"F16"
]
},
"context_length": {
"type": "integer",
"minimum": -9007199254740991,
"maximum": 9007199254740991
},
"model": {
"type": "object",
"properties": {
"id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$"
},
"name": {
"type": "string",
"maxLength": 120
},
"total_params_b": {
"type": "number",
"exclusiveMinimum": 0,
"description": "Total parameters in billions. Sets memory residency."
},
"active_params_b": {
"type": "number",
"exclusiveMinimum": 0,
"description": "Parameters read per token in billions. Equals total for dense models. Sets speed, never memory."
},
"is_moe": {
"type": "boolean"
}
},
"required": [
"id",
"name",
"total_params_b",
"active_params_b",
"is_moe"
],
"additionalProperties": false
},
"hardware": {
"type": "object",
"properties": {
"id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$"
},
"name": {
"type": "string",
"maxLength": 120
},
"memory_gb": {
"type": "number",
"exclusiveMinimum": 0
},
"memory_kind": {
"type": "string",
"enum": [
"dedicated",
"unified",
"shared-system"
]
},
"bandwidth_gb_s": {
"type": "number",
"exclusiveMinimum": 0,
"description": "Theoretical peak memory bandwidth."
},
"status": {
"type": "string",
"enum": [
"current",
"eol",
"used-market",
"announced"
],
"description": "announced means not yet on sale; eol means no longer made."
}
},
"required": [
"id",
"name",
"memory_gb",
"memory_kind",
"bandwidth_gb_s",
"status"
],
"additionalProperties": false
},
"memory": {
"type": "object",
"properties": {
"weights_gb": {
"type": "number",
"minimum": 0,
"description": "All parameters quantised. For a Mixture-of-Experts model this counts every expert."
},
"kv_cache_gb": {
"type": "number",
"minimum": 0
},
"overhead_gb": {
"type": "number",
"minimum": 0,
"description": "Fixed framework overhead."
},
"needed_gb": {
"type": "number",
"minimum": 0,
"description": "weights + KV cache + overhead."
},
"available_gb": {
"type": "number",
"minimum": 0,
"description": "Memory usable for the model after any operating-system reserve."
},
"headroom_gb": {
"type": "number",
"description": "available - needed. Negative when the model does not fit."
},
"offloaded_gb": {
"type": "number",
"minimum": 0,
"description": "Weights that would live in system RAM. 0 when fully resident."
},
"confidence": {
"type": "string",
"enum": [
"measured",
"estimated",
"community"
]
},
"kv_basis": {
"type": "string",
"enum": [
"published_architecture",
"inferred_architecture"
],
"description": "Whether the KV cache is computed from a published config or inferred from the parameter count."
}
},
"required": [
"weights_gb",
"kv_cache_gb",
"overhead_gb",
"needed_gb",
"available_gb",
"headroom_gb",
"offloaded_gb",
"confidence",
"kv_basis"
],
"additionalProperties": false,
"description": "Memory arithmetic for the requested quantisation and context length."
},
"speed": {
"anyOf": [
{
"type": "object",
"properties": {
"low_tok_s": {
"type": "number",
"minimum": 0
},
"high_tok_s": {
"type": "number",
"minimum": 0
},
"confidence": {
"type": "string",
"enum": [
"measured",
"estimated",
"community"
]
},
"basis": {
"anyOf": [
{
"type": "string",
"maxLength": 200
},
{
"type": "null"
}
],
"description": "Where the figure comes from, e.g. a benchmark label. null for a modelled estimate."
},
"sample_count": {
"anyOf": [
{
"type": "integer",
"exclusiveMinimum": 0,
"maximum": 9007199254740991
},
{
"type": "null"
}
],
"description": "Number of measured runs behind a measured or community figure."
},
"last_verified": {
"anyOf": [
{
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$"
},
{
"type": "null"
}
]
}
},
"required": [
"low_tok_s",
"high_tok_s",
"confidence",
"basis",
"sample_count",
"last_verified"
],
"additionalProperties": false,
"description": "Decode speed in tokens per second, always a range with a confidence label."
},
{
"type": "null"
}
],
"description": "null when the model does not fit, because a speed for hardware that cannot load the weights would be invented."
},
"notes": {
"maxItems": 8,
"type": "array",
"items": {
"type": "string",
"maxLength": 300
}
},
"sources": {
"maxItems": 8,
"type": "array",
"items": {
"type": "string",
"pattern": "^https:\\/\\/"
},
"description": "Primary sources for the model and hardware records. May be empty."
},
"url": {
"type": "string",
"pattern": "^https:\\/\\/llmconfigurator\\.com\\/en\\/"
}
},
"required": [
"data_as_of",
"summary",
"verdict",
"quantisation",
"context_length",
"model",
"hardware",
"memory",
"speed",
"notes",
"sources",
"url"
],
"additionalProperties": false
}🟢models_for_hardware(hardware_id, quantisation, context_length, use_case, limit)
Use this when the user asks what models a given machine can run. Needs a hardware id from search_catalog. Returns up to 20 catalogue models ranked for that machine, each with a fit verdict and a speed range, optionally filtered to one use such as coding. It covers only models in the catalogue, at one quantisation and context length per call, and leaves out models that do not fit.
Input Schema
{
"type": "object",
"properties": {
"hardware_id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$",
"description": "Hardware id from search_catalog."
},
"quantisation": {
"default": "Q4_K_M",
"description": "Weight quantisation. One of Q2_K, Q3_K_M, Q4_K_M, Q5_K_M, Q6_K, Q8_0, F16. Default Q4_K_M.",
"type": "string",
"enum": [
"Q2_K",
"Q3_K_M",
"Q4_K_M",
"Q5_K_M",
"Q6_K",
"Q8_0",
"F16"
]
},
"context_length": {
"default": 4096,
"description": "Context window in tokens that the KV cache is sized for. 512 to 262144. Default 4096.",
"type": "integer",
"minimum": 512,
"maximum": 262144
},
"use_case": {
"description": "Only list models tagged for this use. One of coding, chat, reasoning, agents, vision. Omit for all models.",
"type": "string",
"enum": [
"coding",
"chat",
"reasoning",
"agents",
"vision"
]
},
"limit": {
"default": 10,
"description": "Maximum models, 1 to 20. Default 10.",
"type": "integer",
"minimum": 1,
"maximum": 20
}
},
"required": [
"hardware_id"
]
}Output Schema
{
"type": "object",
"properties": {
"data_as_of": {
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$",
"description": "Date the bundled catalogue was last updated. Not the time of this call."
},
"summary": {
"type": "string",
"maxLength": 600,
"description": "One or two plain-language sentences stating the answer."
},
"quantisation": {
"type": "string",
"enum": [
"Q2_K",
"Q3_K_M",
"Q4_K_M",
"Q5_K_M",
"Q6_K",
"Q8_0",
"F16"
]
},
"context_length": {
"type": "integer",
"minimum": -9007199254740991,
"maximum": 9007199254740991
},
"use_case": {
"anyOf": [
{
"type": "string",
"enum": [
"coding",
"chat",
"reasoning",
"agents",
"vision"
]
},
{
"type": "null"
}
],
"description": "The use-case filter applied, or null."
},
"hardware": {
"type": "object",
"properties": {
"id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$"
},
"name": {
"type": "string",
"maxLength": 120
},
"memory_gb": {
"type": "number",
"exclusiveMinimum": 0
},
"memory_kind": {
"type": "string",
"enum": [
"dedicated",
"unified",
"shared-system"
]
},
"bandwidth_gb_s": {
"type": "number",
"exclusiveMinimum": 0,
"description": "Theoretical peak memory bandwidth."
},
"status": {
"type": "string",
"enum": [
"current",
"eol",
"used-market",
"announced"
],
"description": "announced means not yet on sale; eol means no longer made."
}
},
"required": [
"id",
"name",
"memory_gb",
"memory_kind",
"bandwidth_gb_s",
"status"
],
"additionalProperties": false
},
"models": {
"type": "array",
"items": {
"type": "object",
"properties": {
"rank": {
"type": "integer",
"exclusiveMinimum": 0,
"maximum": 9007199254740991
},
"model_id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$"
},
"name": {
"type": "string",
"maxLength": 120
},
"verdict": {
"type": "string",
"enum": [
"fits_comfortably",
"fits_tight",
"needs_offload",
"does_not_fit"
]
},
"needed_gb": {
"type": "number",
"minimum": 0
},
"speed": {
"anyOf": [
{
"type": "object",
"properties": {
"low_tok_s": {
"type": "number",
"minimum": 0
},
"high_tok_s": {
"type": "number",
"minimum": 0
},
"confidence": {
"type": "string",
"enum": [
"measured",
"estimated",
"community"
]
},
"basis": {
"anyOf": [
{
"type": "string",
"maxLength": 200
},
{
"type": "null"
}
],
"description": "Where the figure comes from, e.g. a benchmark label. null for a modelled estimate."
},
"sample_count": {
"anyOf": [
{
"type": "integer",
"exclusiveMinimum": 0,
"maximum": 9007199254740991
},
{
"type": "null"
}
],
"description": "Number of measured runs behind a measured or community figure."
},
"last_verified": {
"anyOf": [
{
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$"
},
{
"type": "null"
}
]
}
},
"required": [
"low_tok_s",
"high_tok_s",
"confidence",
"basis",
"sample_count",
"last_verified"
],
"additionalProperties": false,
"description": "Decode speed in tokens per second, always a range with a confidence label."
},
{
"type": "null"
}
]
},
"url": {
"type": "string",
"pattern": "^https:\\/\\/llmconfigurator\\.com\\/en\\/"
}
},
"required": [
"rank",
"model_id",
"name",
"verdict",
"needed_gb",
"speed",
"url"
],
"additionalProperties": false
}
},
"url": {
"type": "string",
"pattern": "^https:\\/\\/llmconfigurator\\.com\\/en\\/"
}
},
"required": [
"data_as_of",
"summary",
"quantisation",
"context_length",
"use_case",
"hardware",
"models",
"url"
],
"additionalProperties": false
}🟢compare_hardware(hardware_ids)
Use this when the user compares two to four machines. Needs hardware ids from search_catalog. Returns memory, bandwidth, memory type, release year, status and a dated price where one is recorded, side by side. A price of null means none is on record. It does not check any model against the machines; use check_hardware_fit for that.
Input Schema
{
"type": "object",
"properties": {
"hardware_ids": {
"minItems": 2,
"maxItems": 4,
"type": "array",
"items": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$"
},
"description": "Two to four hardware ids from search_catalog."
}
},
"required": [
"hardware_ids"
]
}Output Schema
{
"type": "object",
"properties": {
"data_as_of": {
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$",
"description": "Date the bundled catalogue was last updated. Not the time of this call."
},
"summary": {
"type": "string",
"maxLength": 600,
"description": "One or two plain-language sentences stating the answer."
},
"hardware": {
"minItems": 2,
"maxItems": 4,
"type": "array",
"items": {
"type": "object",
"properties": {
"id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$"
},
"name": {
"type": "string",
"maxLength": 120
},
"memory_gb": {
"type": "number",
"exclusiveMinimum": 0
},
"memory_kind": {
"type": "string",
"enum": [
"dedicated",
"unified",
"shared-system"
]
},
"bandwidth_gb_s": {
"type": "number",
"exclusiveMinimum": 0,
"description": "Theoretical peak memory bandwidth."
},
"status": {
"type": "string",
"enum": [
"current",
"eol",
"used-market",
"announced"
],
"description": "announced means not yet on sale; eol means no longer made."
},
"class": {
"type": "string",
"maxLength": 40
},
"architecture": {
"type": "string",
"maxLength": 80
},
"release_year": {
"anyOf": [
{
"type": "integer",
"minimum": -9007199254740991,
"maximum": 9007199254740991
},
{
"type": "null"
}
]
},
"system_ram_gb": {
"type": "number",
"minimum": 0,
"description": "System RAM available for offload. 0 for unified-memory machines."
},
"price": {
"anyOf": [
{
"type": "object",
"properties": {
"usd": {
"type": "number",
"exclusiveMinimum": 0
},
"as_of": {
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$",
"description": "Date this price was checked."
},
"market": {
"type": "string",
"enum": [
"new",
"used",
"refurb"
]
},
"stale": {
"type": "boolean",
"description": "True when the price is older than 90 days."
},
"source": {
"type": "string",
"maxLength": 200
}
},
"required": [
"usd",
"as_of",
"market",
"stale",
"source"
],
"additionalProperties": false
},
{
"type": "null"
}
],
"description": "null when this catalogue holds no dated price. Never 0."
},
"notes": {
"anyOf": [
{
"type": "string",
"maxLength": 400
},
{
"type": "null"
}
]
},
"sources": {
"maxItems": 8,
"type": "array",
"items": {
"type": "string",
"pattern": "^https:\\/\\/"
}
},
"url": {
"type": "string",
"pattern": "^https:\\/\\/llmconfigurator\\.com\\/en\\/"
}
},
"required": [
"id",
"name",
"memory_gb",
"memory_kind",
"bandwidth_gb_s",
"status",
"class",
"architecture",
"release_year",
"system_ram_gb",
"price",
"notes",
"sources",
"url"
],
"additionalProperties": false
}
}
},
"required": [
"data_as_of",
"summary",
"hardware"
],
"additionalProperties": false
}🟢get_model_specs(model_id)
Use this when the user asks about one model's size, architecture, context window or licence. Needs a model id from search_catalog. Returns total and active parameters, weight size per quantisation, context length, licence, release date and sources. It does not say whether the model fits any machine; use check_hardware_fit for that.
Input Schema
{
"type": "object",
"properties": {
"model_id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$",
"description": "Model id from search_catalog."
}
},
"required": [
"model_id"
]
}Output Schema
{
"type": "object",
"properties": {
"data_as_of": {
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$",
"description": "Date the bundled catalogue was last updated. Not the time of this call."
},
"summary": {
"type": "string",
"maxLength": 600,
"description": "One or two plain-language sentences stating the answer."
},
"model": {
"type": "object",
"properties": {
"id": {
"type": "string",
"minLength": 1,
"maxLength": 80,
"pattern": "^[a-z0-9][a-z0-9._-]*$"
},
"name": {
"type": "string",
"maxLength": 120
},
"total_params_b": {
"type": "number",
"exclusiveMinimum": 0,
"description": "Total parameters in billions. Sets memory residency."
},
"active_params_b": {
"type": "number",
"exclusiveMinimum": 0,
"description": "Parameters read per token in billions. Equals total for dense models. Sets speed, never memory."
},
"is_moe": {
"type": "boolean"
},
"family": {
"type": "string",
"maxLength": 120
},
"context_length": {
"anyOf": [
{
"type": "integer",
"exclusiveMinimum": 0,
"maximum": 9007199254740991
},
{
"type": "null"
}
],
"description": "The model's default context window in tokens."
},
"license": {
"anyOf": [
{
"type": "string",
"maxLength": 80
},
{
"type": "null"
}
]
},
"release_date": {
"anyOf": [
{
"type": "string",
"pattern": "^\\d{4}-\\d{2}-\\d{2}$"
},
{
"type": "null"
}
]
},
"status": {
"type": "string",
"enum": [
"current",
"legacy",
"deprecated"
]
}
},
"required": [
"id",
"name",
"total_params_b",
"active_params_b",
"is_moe",
"family",
"context_length",
"license",
"release_date",
"status"
],
"additionalProperties": false
},
"quantisations": {
"type": "array",
"items": {
"type": "object",
"properties": {
"quantisation": {
"type": "string",
"enum": [
"Q2_K",
"Q3_K_M",
"Q4_K_M",
"Q5_K_M",
"Q6_K",
"Q8_0",
"F16"
]
},
"bits_per_weight": {
"type": "number",
"exclusiveMinimum": 0
},
"weights_gb": {
"type": "number",
"minimum": 0
},
"confidence": {
"type": "string",
"enum": [
"measured",
"estimated",
"community"
]
}
},
"required": [
"quantisation",
"bits_per_weight",
"weights_gb",
"confidence"
],
"additionalProperties": false
},
"description": "Weight size per quantisation. Excludes KV cache and overhead, which depend on context length."
},
"sources": {
"maxItems": 8,
"type": "array",
"items": {
"type": "string",
"pattern": "^https:\\/\\/"
}
},
"url": {
"type": "string",
"pattern": "^https:\\/\\/llmconfigurator\\.com\\/en\\/"
}
},
"required": [
"data_as_of",
"summary",
"model",
"quantisations",
"sources",
"url"
],
"additionalProperties": false
}Community
Evidence