Run models locally as a first-class provider. The CLI grows a managed llama.cpp runtime (engine install, model download, server supervision); the desktop app grows the full setup and management story on top of it. GUI surfaces ship behind the desktop --local launch flag (hermes desktop --local, or the flag on the packaged app); backend routes and the CLI are always live. Runtime (hermes_cli/local_runtime/): - curated GGUF catalog with per-machine variant selection: hardware probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice by context window - derived recommendation: quality-ranked picks gated by a predicted decode-speed floor, bandwidth-aware on unified memory; the decision table is pinned as a test (pick AND reason per memory class), and the Recommended badge explains its pick in a tooltip fed by the resolver's actual branch - engine install + model download with resumable split parts, cumulative plan-level progress, and staged-model integrity (a split GGUF counts only when every part is present) - server supervision: spawn/adopt/stop, router mode with per-model load progress relayed over SSE, abandoned-request cleanup Desktop: - Settings -> Providers -> Local models: one-click quickstart (install engine, download the recommended model, boot) plus per-model download/ activate/eject, fit-ranked catalog with context pills - model pickers (composer dropdown + Cmd+K) show staged local models, in-flight downloads as live progress rows, and load-into-memory bars - local-setup campaign tip for eligible hardware; System resources statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends - friendly dead-server errors, and failed agent builds retry on the next send instead of wedging the session Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
175 lines
4.8 KiB
JSON
175 lines
4.8 KiB
JSON
{
|
|
"schema_version": 1,
|
|
"models": [
|
|
{
|
|
"id": "qwen3.8-27b",
|
|
"display_name": "Qwen3.8 27B",
|
|
"description": "Best all-round agent model; sees images; long context stays fast",
|
|
"repo": "unsloth/Qwen3.8-27B-GGUF",
|
|
"variants": [
|
|
{
|
|
"quant": "UD-Q4_K_M",
|
|
"files": [
|
|
{
|
|
"path": "Qwen3.8-27B-UD-Q4_K_M.gguf",
|
|
"size_bytes": 16464440224
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"n_ctx_train": 262144,
|
|
"full_layers": 16,
|
|
"recurrent_layers": 48,
|
|
"per_layer_f16": 4096,
|
|
"n_vocab": 248320,
|
|
"mmproj": {
|
|
"path": "mmproj-BF16.gguf",
|
|
"size_bytes": 931146432,
|
|
"local": "mmproj-Qwen3.8-27B-BF16.gguf"
|
|
},
|
|
"mtp": true,
|
|
"mtp_draft_depth": 2,
|
|
"sampling": {
|
|
"temp": "1.0",
|
|
"top-p": "0.95",
|
|
"top-k": "20",
|
|
"min-p": "0.0"
|
|
},
|
|
"quality": 90,
|
|
"decode_fraction": 1.0
|
|
},
|
|
{
|
|
"id": "qwen3.8-flash-next",
|
|
"display_name": "Qwen3.8 Flash Next",
|
|
"description": "Frontier-scale model; needs a very large GPU to run well",
|
|
"repo": "unsloth/Qwen3.8-Flash-Next-GGUF",
|
|
"variants": [
|
|
{
|
|
"quant": "UD-Q4_K_XL",
|
|
"files": [
|
|
{
|
|
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf",
|
|
"size_bytes": 10946624
|
|
},
|
|
{
|
|
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00002-of-00004.gguf",
|
|
"size_bytes": 49859583136
|
|
},
|
|
{
|
|
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00003-of-00004.gguf",
|
|
"size_bytes": 49376141504
|
|
},
|
|
{
|
|
"path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00004-of-00004.gguf",
|
|
"size_bytes": 12087983520
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"n_ctx_train": 262144,
|
|
"full_layers": 12,
|
|
"recurrent_layers": 36,
|
|
"per_layer_f16": 2048,
|
|
"moe": true,
|
|
"n_vocab": 248320,
|
|
"mmproj": {
|
|
"path": "mmproj-BF16.gguf",
|
|
"size_bytes": 907542944,
|
|
"local": "mmproj-Qwen3.8-Flash-Next-BF16.gguf"
|
|
},
|
|
"min_engine": "b10678",
|
|
"quality": 95,
|
|
"decode_fraction": 0.08
|
|
},
|
|
{
|
|
"id": "qwen3.6-35b-a3b",
|
|
"display_name": "Qwen3.6 35B-A3B",
|
|
"description": "Bigger mixture-of-experts with multi-token prediction; sees images",
|
|
"repo": "unsloth/Qwen3.6-35B-A3B-MTP-GGUF",
|
|
"variants": [
|
|
{
|
|
"quant": "UD-Q4_K_M",
|
|
"files": [
|
|
{
|
|
"path": "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf",
|
|
"size_bytes": 22663387424
|
|
}
|
|
],
|
|
"validated": true
|
|
}
|
|
],
|
|
"n_ctx_train": 262144,
|
|
"full_layers": 10,
|
|
"recurrent_layers": 30,
|
|
"per_layer_f16": 2048,
|
|
"moe": true,
|
|
"mtp": true,
|
|
"n_vocab": 248320,
|
|
"mtp_draft_depth": 2,
|
|
"mmproj": {
|
|
"path": "mmproj-BF16.gguf",
|
|
"size_bytes": 902822528,
|
|
"local": "mmproj-Qwen3.6-35B-A3B-BF16.gguf"
|
|
},
|
|
"sampling": {
|
|
"temp": "1.0",
|
|
"top-p": "0.95",
|
|
"top-k": "20",
|
|
"min-p": "0.0"
|
|
},
|
|
"quality": 80,
|
|
"decode_fraction": 0.15
|
|
},
|
|
{
|
|
"id": "deepseek-v4-flash",
|
|
"display_name": "DeepSeek V4 Flash",
|
|
"description": "Frontier-class model for machines with 128GB+ memory",
|
|
"repo": "unsloth/DeepSeek-V4-Flash-0731-GGUF",
|
|
"variants": [
|
|
{
|
|
"quant": "UD-Q4_K_XL",
|
|
"files": [
|
|
{
|
|
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00001-of-00005.gguf",
|
|
"size_bytes": 5257408
|
|
},
|
|
{
|
|
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00002-of-00005.gguf",
|
|
"size_bytes": 48935523072
|
|
},
|
|
{
|
|
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00003-of-00005.gguf",
|
|
"size_bytes": 48980787136
|
|
},
|
|
{
|
|
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00004-of-00005.gguf",
|
|
"size_bytes": 49999168416
|
|
},
|
|
{
|
|
"path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00005-of-00005.gguf",
|
|
"size_bytes": 7174505088
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"n_ctx_train": 1048576,
|
|
"full_layers": 43,
|
|
"recurrent_layers": 0,
|
|
"per_layer_f16": 1152,
|
|
"moe": true,
|
|
"n_vocab": 163840,
|
|
"draft": {
|
|
"path": "dspark-DeepSeek-V4-Flash-0731-Q8_0.gguf",
|
|
"size_bytes": 10896057440
|
|
},
|
|
"sampling": {
|
|
"temp": "1.0",
|
|
"top-p": "0.95",
|
|
"min-p": "0.01"
|
|
},
|
|
"quality": 85,
|
|
"decode_fraction": 0.1
|
|
}
|
|
]
|
|
}
|