{"$schema":"https://wellknown.network/schemas/agent-record-v1.json","schemaVersion":"1","id":"ag_nntkspr9k2d6","handle":"nodegrove-vram-can-i-run-it","url":"https://wellknown.network/agents/nodegrove-vram-can-i-run-it","links":{"self":"https://wellknown.network/agents/nodegrove-vram-can-i-run-it/record.json","html":"https://wellknown.network/agents/nodegrove-vram-can-i-run-it","markdown":"https://wellknown.network/agents/nodegrove-vram-can-i-run-it/record.md","api":"https://wellknown.network/api/v1/agents/nodegrove-vram-can-i-run-it","status":"https://wellknown.network/api/v1/agents/nodegrove-vram-can-i-run-it/status","claim":"https://wellknown.network/agents/nodegrove-vram-can-i-run-it/claim","claimApi":"https://wellknown.network/api/v1/claims","claimDescriptor":"https://wellknown.network/agents/nodegrove-vram-can-i-run-it/claim.json","badge":"https://wellknown.network/agents/nodegrove-vram-can-i-run-it/badge.svg","openapi":"https://wellknown.network/openapi.json","history":"https://wellknown.network/api/v1/agents/nodegrove-vram-can-i-run-it/history","tools":"https://wellknown.network/api/v1/agents/nodegrove-vram-can-i-run-it/tools"},"ard":{"identifier":"urn:air:mcp.nodegrove.io:server:nodegrove-vram-can-i-run-it","type":"application/mcp-server-card+json"},"kind":"mcp_server","declared":{"name":"Nodegrove VRAM: can I run it?","summary":"Can this LLM run on my GPU? VRAM, speed ceiling and what fits instead, for any model and GPU.","description":"Can this LLM run on my GPU? VRAM, speed ceiling and what fits instead, for any model and GPU.","publisher":{"name":"io.nodegrove","url":null},"homepage":"https://nodegrove.io/mcp","repository":"https://github.com/nodegrove/vram-mcp","version":"1.1.1","license":"(MIT AND CC-BY-4.0)","protocols":["mcp"],"tags":["mcp","model-context-protocol","mcp-server","llm","vram","gpu","local-llm","hugging-face","kv-cache"],"pricing":null,"endpoints":[{"url":"https://mcp.nodegrove.io/mcp","type":"mcp_streamable_http","auth":null,"probeable":true},{"url":"npm:@nodegrove/vram-mcp","type":"package_npm","auth":null,"probeable":false}],"skills":null,"tools":null,"extra":{"updatedAt":"2026-10-05T22:16:32.353566Z","npmKeywords":["mcp","model-context-protocol","mcp-server","llm","vram","gpu","local-llm","hugging-face","kv-cache"],"publishedAt":"2026-10-05T22:16:32.353566Z","registryName":"io.nodegrove/vram-mcp"},"attribution":{"kind":"mcp_registry","name":"mcp_registry","license":"npm","repoUrl":"mcp_registry","summary":"mcp_registry","version":"mcp_registry","description":"mcp_registry","homepageUrl":"mcp_registry","publisherName":"mcp_registry"}},"derived":{"capabilities":[{"slug":"ai.model-access","name":"Model Access","confidence":1,"provenance":"derived"},{"slug":"commerce.pricing","name":"Pricing & Quotes","confidence":0.525,"provenance":"derived"}],"categories":["ai","commerce"],"language":"en"},"observed":{"status":"live","statusReason":"Responded 2h ago.","lastOkAt":"2026-10-10T14:27:03.021Z","lastProbedAt":"2026-10-10T14:27:03.021Z","statusComputedAt":"2026-10-10T14:27:33.846Z","reliability30d":{"probes":19,"successRate":1,"p50Ms":45,"basis":"service","measures":{"availability":"availability","latency":"response time","tools":"tool surface observed","summary":"Checks reached the service itself."},"checks":{"total":19,"ok":19,"authBoundaryOk":0,"serviceOk":19,"note":"Counted from the observation rows for the window, checks of the server only (HTTP, A2A card, MCP initialize). ok = authBoundaryOk + serviceOk. `probes` is the sum of daily rollups and includes registry checks, so it can differ from `total`."}},"latestObservations":[{"at":"2026-10-10T14:27:03.021Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":31,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-09): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"23026f28e63c1f752549df1b1b94e93caccdf85b8d596c7ea6377d4fac8a64ac","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.2","protocolVersion":"2025-06-18"}},{"at":"2026-10-10T08:32:49.635Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":35,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-09): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"23026f28e63c1f752549df1b1b94e93caccdf85b8d596c7ea6377d4fac8a64ac","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.2","protocolVersion":"2025-06-18"}},{"at":"2026-10-10T01:24:03.600Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":39,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-09): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"23026f28e63c1f752549df1b1b94e93caccdf85b8d596c7ea6377d4fac8a64ac","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.2","protocolVersion":"2025-06-18"}},{"at":"2026-10-09T18:25:53.021Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":40,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-09): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"23026f28e63c1f752549df1b1b94e93caccdf85b8d596c7ea6377d4fac8a64ac","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.2","protocolVersion":"2025-06-18"}},{"at":"2026-10-09T11:25:50.615Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":35,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-09): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"23026f28e63c1f752549df1b1b94e93caccdf85b8d596c7ea6377d4fac8a64ac","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.2","protocolVersion":"2025-06-18"}},{"at":"2026-10-09T04:27:57.832Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":38,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-09): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"23026f28e63c1f752549df1b1b94e93caccdf85b8d596c7ea6377d4fac8a64ac","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.2","protocolVersion":"2025-06-18"}},{"at":"2026-10-08T21:22:04.996Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":38,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-06): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"596ce8330371256675b46e7706758028ec0a7d0f645ee8143f3b7c4f17404f94","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.1","protocolVersion":"2025-06-18"}},{"at":"2026-10-08T14:23:07.922Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":94,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-06): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"596ce8330371256675b46e7706758028ec0a7d0f645ee8143f3b7c4f17404f94","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.1","protocolVersion":"2025-06-18"}},{"at":"2026-10-08T08:29:01.874Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":58,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-06): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"596ce8330371256675b46e7706758028ec0a7d0f645ee8143f3b7c4f17404f94","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.1","protocolVersion":"2025-06-18"}},{"at":"2026-10-08T01:21:01.658Z","kind":"mcp_initialize","ok":true,"httpStatus":200,"latencyMs":45,"error":null,"detail":{"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-06): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"toolCount":6,"toolsHash":"596ce8330371256675b46e7706758028ec0a7d0f645ee8143f3b7c4f17404f94","serverName":"nodegrove-vram","capabilities":["tools"],"serverVersion":"1.1.1","protocolVersion":"2025-06-18"}}],"tools":[{"name":"can_i_run","description":"Can this GPU run this open-weight LLM? Returns fits, tight or no, the memory split (weights, KV cache, overhead), a decode-speed ceiling, the longest context th"},{"name":"what_fits","description":"Which open-weight LLMs fit this GPU: every model in list_models checked at one quantisation and context, with a recommended everyday model (the biggest class th"},{"name":"estimate_vram","description":"How much memory an LLM needs: weights + KV cache + overhead at each quantisation (or one), at a given context, and the smallest common card class that holds eac"},{"name":"estimate_from_hf_repo","description":"Reads any Hugging Face model repo's config.json and parameter count and estimates its memory: the attention layout found (standard, sliding-window, hybrid or la"},{"name":"list_models","description":"The open-weight LLMs nodegrove.io has verified against their config.json (data version 2026-10-09): id, size, attention design, native context, licence, memory "},{"name":"list_gpus","description":"The GPUs and machines nodegrove.io covers: memory, the memory a runtime can use and bandwidth, from the makers' specs, with each one's page."}],"package":{"name":"@nodegrove/vram-mcp","registry":"npm","observedAt":"2026-10-10T14:28:40.707Z","publishedAt":"2026-10-09T00:35:29.554Z","latestVersion":"1.1.2","weeklyDownloads":445},"toolSurface":{"id":"ts_d78vqdnq647x","endpointId":"ep_3rtc584ry4jf","hash":"23026f28e63c1f752549df1b1b94e93caccdf85b8d596c7ea6377d4fac8a64ac","toolCount":6,"serverName":"nodegrove-vram","serverVersion":"1.1.2","protocolVersion":"2025-06-18","firstSeenAt":"2026-10-09T04:27:57.825Z","lastSeenAt":"2026-10-10T14:27:03.007Z","observations":6,"toolNames":["can_i_run","what_fits","estimate_vram","estimate_from_hf_repo","list_models","list_gpus"],"distinctSurfaces":3},"endpointFacts":[{"id":"ep_3rtc584ry4jf","url":"https://mcp.nodegrove.io/mcp","type":"mcp_streamable_http","factsCheckedAt":"2026-10-09T11:25:50.632Z","auth":{"observedAt":"2026-10-09T11:25:50.647Z","authRequired":false,"scheme":null,"resourceMetadata":null,"authorizationServer":null,"conformance":{"dpop":false,"rfc8414":false,"rfc9728":false,"pkceS256":false,"clientIdMetadataDocument":false,"dynamicClientRegistration":false}},"tls":{"observedAt":"2026-10-09T11:25:50.657Z","protocol":"TLSv1.3","chainValid":true,"chainError":null,"hostMatches":true,"subject":"nodegrove.io","issuer":{"commonName":"WE1","organization":"Google Trust Services"},"validFrom":"2026-10-02T16:39:39.000Z","validTo":"2026-12-31T17:39:32.000Z","daysToExpiry":82,"sanCount":3,"fingerprint256":"35:8C:9F:59:AC:04:C5:F1:67:92:CF:98:C1:78:A1:8E:F6:7C:3F:1E:14:67:AB:98:BF:7A:18:43:C3:08:A6:76"}}]},"verification":{"claimed":false,"claimedAt":null,"proofs":[]},"provenance":{"sources":[{"source":"npm","key":"@nodegrove/vram-mcp","url":"https://www.npmjs.com/package/@nodegrove/vram-mcp","firstSeenAt":"2026-10-05T23:18:54.417Z","fetchedAt":"2026-10-09T23:21:57.175Z","normalizedAt":"2026-10-09T23:21:57.175Z"},{"source":"mcp_registry","key":"io.nodegrove/vram-mcp","url":"https://registry.modelcontextprotocol.io/v0/servers/io.nodegrove%2Fvram-mcp","firstSeenAt":"2026-10-05T16:18:49.056Z","fetchedAt":"2026-10-08T17:21:38.153Z","normalizedAt":"2026-10-08T17:21:38.153Z"}]},"firstSeenAt":"2026-10-05T16:18:49.056Z","updatedAt":"2026-10-10T14:28:40.707Z"}